mirror of
https://github.com/opendatalab/MinerU.git
synced 2026-09-24 23:10:23 +08:00
Merge pull request #5063 from opendatalab/dev
This commit is contained in:
@@ -21,6 +21,10 @@ Options:
|
||||
-e, --end INTEGER Ending page number for parsing (0-based)
|
||||
-f, --formula BOOLEAN Enable formula parsing (default: enabled)
|
||||
-t, --table BOOLEAN Enable table parsing (default: enabled)
|
||||
--client-side-output-generation BOOLEAN
|
||||
Generate Markdown and content lists locally
|
||||
from server-returned middle JSON, images, and
|
||||
original files (default: disabled)
|
||||
--help Show help information
|
||||
```
|
||||
> [!TIP]
|
||||
@@ -59,6 +63,9 @@ Options:
|
||||
starts a reusable local mineru-api service.
|
||||
--enable-vlm-preload BOOLEAN Preload the local VLM model when gradio
|
||||
starts a local mineru-api service.
|
||||
--client-side-output-generation BOOLEAN
|
||||
Generate Markdown and content lists locally
|
||||
from server-returned middle JSON.
|
||||
--latex-delimiters-type [a|b|all]
|
||||
Set the type of LaTeX delimiters to use in
|
||||
Markdown rendering: 'a' for type '$', 'b' for
|
||||
|
||||
@@ -21,6 +21,9 @@ Options:
|
||||
-e, --end INTEGER 结束解析的页码(从 0 开始)
|
||||
-f, --formula BOOLEAN 是否启用公式解析(默认开启)
|
||||
-t, --table BOOLEAN 是否启用表格解析(默认开启)
|
||||
--client-side-output-generation BOOLEAN
|
||||
在客户端基于服务端返回的 middle JSON、图片与原文件
|
||||
生成 Markdown 和 content list(默认关闭)
|
||||
--help 显示帮助信息
|
||||
```
|
||||
> [!TIP]
|
||||
@@ -54,6 +57,9 @@ Options:
|
||||
mineru-api
|
||||
--enable-vlm-preload BOOLEAN 在 Gradio 拉起本地 mineru-api 时预加载本地
|
||||
VLM 模型
|
||||
--client-side-output-generation BOOLEAN
|
||||
在客户端基于服务端返回的 middle JSON 生成 Markdown
|
||||
和 content list
|
||||
--latex-delimiters-type [a|b|all]
|
||||
设置在 Markdown 渲染中使用的 LaTeX 分隔符类型
|
||||
('a' 表示 '$' 类型,'b' 表示 '()[]' 类型,
|
||||
|
||||
@@ -19,7 +19,13 @@ from mineru.backend.hybrid.hybrid_model_output_to_middle_json import (
|
||||
init_middle_json,
|
||||
)
|
||||
from mineru.backend.utils.runtime_utils import exclude_progress_bar_idle_time
|
||||
from mineru.backend.pipeline.model_init import HybridModelSingleton
|
||||
from mineru.backend.pipeline.model_init import (
|
||||
HybridModelSingleton,
|
||||
run_layout_inference,
|
||||
run_mfr_inference,
|
||||
run_ocr_det_inference,
|
||||
run_ocr_rec_inference,
|
||||
)
|
||||
from mineru.backend.vlm.vlm_analyze import (
|
||||
ModelSingleton,
|
||||
aio_predictor_execution_guard,
|
||||
@@ -119,8 +125,11 @@ def ocr_det(
|
||||
page_mfd_res, useful_list
|
||||
)
|
||||
bgr_image = cv2.cvtColor(new_image, cv2.COLOR_RGB2BGR)
|
||||
ocr_res = hybrid_pipeline_model.ocr_model.ocr(
|
||||
bgr_image, mfd_res=adjusted_mfdetrec_res, rec=False
|
||||
ocr_res = run_ocr_det_inference(
|
||||
hybrid_pipeline_model.ocr_model.ocr,
|
||||
bgr_image,
|
||||
mfd_res=adjusted_mfdetrec_res,
|
||||
rec=False,
|
||||
)[0]
|
||||
if ocr_res:
|
||||
ocr_result_list = get_ocr_result_list(
|
||||
@@ -193,7 +202,11 @@ def ocr_det(
|
||||
|
||||
# 批处理检测
|
||||
det_batch_size = min(len(batch_images), batch_ratio * OCR_DET_BASE_BATCH_SIZE)
|
||||
batch_results = hybrid_pipeline_model.ocr_model.text_detector.batch_predict(batch_images, det_batch_size)
|
||||
batch_results = run_ocr_det_inference(
|
||||
hybrid_pipeline_model.ocr_model.text_detector.batch_predict,
|
||||
batch_images,
|
||||
det_batch_size,
|
||||
)
|
||||
|
||||
# 处理批处理结果
|
||||
for crop_info, (dt_boxes, _) in zip(group_crops, batch_results):
|
||||
@@ -392,7 +405,8 @@ def _predict_layout_for_title_split(
|
||||
batch_ratio,
|
||||
):
|
||||
"""执行layout小模型检测,专门为Hybrid标题拆分提供页面layout结果。"""
|
||||
return hybrid_pipeline_model.layout_model.batch_predict(
|
||||
return run_layout_inference(
|
||||
hybrid_pipeline_model.layout_model.batch_predict,
|
||||
images,
|
||||
batch_size=min(8, batch_ratio * LAYOUT_BASE_BATCH_SIZE),
|
||||
)
|
||||
@@ -433,7 +447,8 @@ def _process_ocr_and_formulas(
|
||||
if inline_formula_enable:
|
||||
images_mfd_res = _build_inline_formula_inputs(images_layout_res)
|
||||
# 公式识别
|
||||
inline_formula_list = hybrid_pipeline_model.mfr_model.batch_predict(
|
||||
inline_formula_list = run_mfr_inference(
|
||||
hybrid_pipeline_model.mfr_model.batch_predict,
|
||||
images_mfd_res,
|
||||
np_images,
|
||||
batch_size=batch_ratio * MFR_BASE_BATCH_SIZE,
|
||||
@@ -473,7 +488,12 @@ def _process_ocr_and_formulas(
|
||||
img_crop_list.append(ocr_res.pop('np_img'))
|
||||
if len(img_crop_list) > 0:
|
||||
# Process OCR
|
||||
ocr_result_list = hybrid_pipeline_model.ocr_model.ocr(img_crop_list, det=False, tqdm_enable=True)[0]
|
||||
ocr_result_list = run_ocr_rec_inference(
|
||||
hybrid_pipeline_model.ocr_model.ocr,
|
||||
img_crop_list,
|
||||
det=False,
|
||||
tqdm_enable=True,
|
||||
)[0]
|
||||
|
||||
# Verify we have matching counts
|
||||
assert len(ocr_result_list) == len(need_ocr_list), f'ocr_result_list: {len(ocr_result_list)}, need_ocr_list: {len(need_ocr_list)}'
|
||||
|
||||
@@ -14,13 +14,14 @@ from mineru.backend.utils.para_block_utils import (
|
||||
)
|
||||
from mineru.backend.hybrid.hybrid_magic_model import MagicModel
|
||||
from mineru.backend.utils.runtime_utils import cross_page_table_merge
|
||||
from mineru.backend.pipeline.model_init import run_ocr_rec_inference
|
||||
from mineru.utils.config_reader import get_table_enable
|
||||
from mineru.utils.cut_image import cut_image_and_table
|
||||
from mineru.utils.enum_class import ContentType, BlockType
|
||||
from mineru.utils.hash_utils import bytes_md5
|
||||
from mineru.utils.ocr_utils import OcrConfidence, rotate_vertical_crop_if_needed
|
||||
from mineru.utils.title_level_postprocess import apply_title_leveling_to_pdf_info
|
||||
from mineru.utils.pdfium_guard import close_pdfium_document, pdfium_guard
|
||||
from mineru.utils.pdfium_guard import close_pdfium_child, close_pdfium_document, pdfium_guard
|
||||
from mineru.version import __version__
|
||||
|
||||
|
||||
@@ -138,7 +139,12 @@ def _apply_post_ocr(pdf_info_list, hybrid_pipeline_model):
|
||||
img_crop_list.append(rotate_vertical_crop_if_needed(span['np_img']))
|
||||
span.pop('np_img')
|
||||
if len(img_crop_list) > 0:
|
||||
ocr_res_list = hybrid_pipeline_model.ocr_model.ocr(img_crop_list, det=False, tqdm_enable=True)[0]
|
||||
ocr_res_list = run_ocr_rec_inference(
|
||||
hybrid_pipeline_model.ocr_model.ocr,
|
||||
img_crop_list,
|
||||
det=False,
|
||||
tqdm_enable=True,
|
||||
)[0]
|
||||
assert len(ocr_res_list) == len(
|
||||
need_ocr_list), f'ocr_res_list: {len(ocr_res_list)}, need_ocr_list: {len(need_ocr_list)}'
|
||||
for index, span in enumerate(need_ocr_list):
|
||||
@@ -192,17 +198,21 @@ def append_page_results_to_middle_json(
|
||||
zip(model_list, images_list)
|
||||
):
|
||||
page_index = page_start_index + offset
|
||||
with pdfium_guard():
|
||||
page = pdf_doc[page_index]
|
||||
page_info = blocks_to_page_info(
|
||||
page_model_list,
|
||||
image_dict,
|
||||
page,
|
||||
image_writer,
|
||||
page_index,
|
||||
_ocr_enable,
|
||||
_vlm_ocr_enable,
|
||||
)
|
||||
page = None
|
||||
try:
|
||||
with pdfium_guard():
|
||||
page = pdf_doc[page_index]
|
||||
page_info = blocks_to_page_info(
|
||||
page_model_list,
|
||||
image_dict,
|
||||
page,
|
||||
image_writer,
|
||||
page_index,
|
||||
_ocr_enable,
|
||||
_vlm_ocr_enable,
|
||||
)
|
||||
finally:
|
||||
close_pdfium_child(page)
|
||||
middle_json["pdf_info"].append(page_info)
|
||||
if progress_bar is not None:
|
||||
progress_bar.update(1)
|
||||
|
||||
@@ -9,7 +9,13 @@ from tqdm import tqdm
|
||||
from collections import defaultdict
|
||||
import numpy as np
|
||||
|
||||
from .model_init import AtomModelSingleton
|
||||
from .model_init import (
|
||||
AtomModelSingleton,
|
||||
run_layout_inference,
|
||||
run_mfr_inference,
|
||||
run_ocr_det_inference,
|
||||
run_ocr_rec_inference,
|
||||
)
|
||||
from .model_list import AtomicModel
|
||||
from ...utils.config_reader import (
|
||||
get_formula_enable,
|
||||
@@ -362,9 +368,10 @@ class BatchAnalyze:
|
||||
np_images = [np.asarray(image) for image, _, _ in images_with_extra_info]
|
||||
|
||||
# pp-doclayout_v2
|
||||
images_layout_res += self.model.layout_model.batch_predict(
|
||||
images_layout_res += run_layout_inference(
|
||||
self.model.layout_model.batch_predict,
|
||||
pil_images,
|
||||
batch_size=min(8, self.batch_ratio * LAYOUT_BASE_BATCH_SIZE)
|
||||
batch_size=min(8, self.batch_ratio * LAYOUT_BASE_BATCH_SIZE),
|
||||
)
|
||||
# 清理显存
|
||||
clean_vram(self.model.device, vram_threshold=8)
|
||||
@@ -380,7 +387,8 @@ class BatchAnalyze:
|
||||
images_mfd_res.append(page_formula_res)
|
||||
|
||||
# 公式识别
|
||||
images_formula_list = self.model.mfr_model.batch_predict(
|
||||
images_formula_list = run_mfr_inference(
|
||||
self.model.mfr_model.batch_predict,
|
||||
images_mfd_res,
|
||||
np_images,
|
||||
batch_size=self.batch_ratio * MFR_BASE_BATCH_SIZE,
|
||||
@@ -526,7 +534,9 @@ class BatchAnalyze:
|
||||
if inline_mask_boxes
|
||||
else bgr_image
|
||||
)
|
||||
ocr_result = det_ocr_engine.ocr(det_image, rec=False)[0]
|
||||
ocr_result = run_ocr_det_inference(
|
||||
det_ocr_engine.ocr, det_image, rec=False
|
||||
)[0]
|
||||
if ocr_result and formula_mask_boxes:
|
||||
ocr_result = update_det_boxes(ocr_result, formula_mask_boxes)
|
||||
if ocr_result:
|
||||
@@ -555,7 +565,13 @@ class BatchAnalyze:
|
||||
enable_merge_det_boxes=False,
|
||||
)
|
||||
cropped_img_list = [item["cropped_img"] for item in rec_img_list]
|
||||
ocr_res_list = ocr_engine.ocr(cropped_img_list, det=False, tqdm_enable=True, tqdm_desc=f"Table-ocr rec {_lang}")[0]
|
||||
ocr_res_list = run_ocr_rec_inference(
|
||||
ocr_engine.ocr,
|
||||
cropped_img_list,
|
||||
det=False,
|
||||
tqdm_enable=True,
|
||||
tqdm_desc=f"Table-ocr rec {_lang}",
|
||||
)[0]
|
||||
# 按照 table_id 将识别结果进行回填
|
||||
for img_dict, ocr_res in zip(rec_img_list, ocr_res_list):
|
||||
ocr_text = self._normalize_table_ocr_rec_text(ocr_res[0])
|
||||
@@ -715,7 +731,9 @@ class BatchAnalyze:
|
||||
|
||||
# 批处理检测
|
||||
det_batch_size = min(len(batch_images), self.batch_ratio * OCR_DET_BASE_BATCH_SIZE)
|
||||
batch_results = ocr_model.text_detector.batch_predict(batch_images, det_batch_size)
|
||||
batch_results = run_ocr_det_inference(
|
||||
ocr_model.text_detector.batch_predict, batch_images, det_batch_size
|
||||
)
|
||||
|
||||
# 处理批处理结果
|
||||
for crop_info, (dt_boxes, _) in zip(group_crops, batch_results):
|
||||
@@ -775,8 +793,11 @@ class BatchAnalyze:
|
||||
bgr_image,
|
||||
adjusted_mfdetrec_res,
|
||||
)
|
||||
ocr_res = ocr_model.ocr(
|
||||
det_image, mfd_res=adjusted_mfdetrec_res, rec=False
|
||||
ocr_res = run_ocr_det_inference(
|
||||
ocr_model.ocr,
|
||||
det_image,
|
||||
mfd_res=adjusted_mfdetrec_res,
|
||||
rec=False,
|
||||
)[0]
|
||||
|
||||
# Integration results
|
||||
@@ -831,7 +852,9 @@ class BatchAnalyze:
|
||||
atom_model_name=AtomicModel.OCR,
|
||||
lang=lang
|
||||
)
|
||||
ocr_res_list = ocr_model.ocr(img_crop_list, det=False, tqdm_enable=True)[0]
|
||||
ocr_res_list = run_ocr_rec_inference(
|
||||
ocr_model.ocr, img_crop_list, det=False, tqdm_enable=True
|
||||
)[0]
|
||||
|
||||
# Verify we have matching counts
|
||||
assert len(ocr_res_list) == len(
|
||||
@@ -900,7 +923,9 @@ class BatchAnalyze:
|
||||
)
|
||||
|
||||
seal_crop_bgr = cv2.cvtColor(seal_crop_rgb, cv2.COLOR_RGB2BGR)
|
||||
seal_ocr_res = seal_ocr_model.ocr(seal_crop_bgr, det=True, rec=True)[0]
|
||||
seal_ocr_res = run_ocr_det_inference(
|
||||
seal_ocr_model.ocr, seal_crop_bgr, det=True, rec=True
|
||||
)[0]
|
||||
if not seal_ocr_res:
|
||||
continue
|
||||
|
||||
|
||||
@@ -19,6 +19,52 @@ from ...utils.enum_class import ModelPath
|
||||
from ...utils.models_download_utils import auto_download_and_get_model_root_path
|
||||
|
||||
PIPELINE_MODEL_INIT_LOCK = threading.RLock()
|
||||
# 这些锁保护 pipeline 与 hybrid 共享的 atom model/native 模型推理调用,避免多线程同时进入同一个模型对象。
|
||||
PIPELINE_LAYOUT_INFERENCE_LOCK = threading.RLock()
|
||||
PIPELINE_MFR_INFERENCE_LOCK = threading.RLock()
|
||||
PIPELINE_OCR_DET_INFERENCE_LOCK = threading.RLock()
|
||||
PIPELINE_OCR_REC_INFERENCE_LOCK = threading.RLock()
|
||||
# 临时关闭 pipeline/hybrid 共享推理阶段锁;需要回滚实验时可通过环境变量重新打开。
|
||||
PIPELINE_INFERENCE_LOCKS_ENABLED = os.getenv(
|
||||
'MINERU_ENABLE_PIPELINE_INFERENCE_LOCKS', 'False'
|
||||
).lower() in ['true', '1', 'yes']
|
||||
|
||||
|
||||
def _run_with_inference_lock(inference_lock, inference_callable, *args, **kwargs):
|
||||
"""按实验开关决定是否在指定推理锁内执行真实 native 模型调用。"""
|
||||
if not PIPELINE_INFERENCE_LOCKS_ENABLED:
|
||||
return inference_callable(*args, **kwargs)
|
||||
|
||||
with inference_lock:
|
||||
return inference_callable(*args, **kwargs)
|
||||
|
||||
|
||||
def run_layout_inference(inference_callable, *args, **kwargs):
|
||||
"""按实验开关执行共享 Layout 模型调用。"""
|
||||
return _run_with_inference_lock(
|
||||
PIPELINE_LAYOUT_INFERENCE_LOCK, inference_callable, *args, **kwargs
|
||||
)
|
||||
|
||||
|
||||
def run_mfr_inference(inference_callable, *args, **kwargs):
|
||||
"""按实验开关执行共享 MFR 模型调用。"""
|
||||
return _run_with_inference_lock(
|
||||
PIPELINE_MFR_INFERENCE_LOCK, inference_callable, *args, **kwargs
|
||||
)
|
||||
|
||||
|
||||
def run_ocr_det_inference(inference_callable, *args, **kwargs):
|
||||
"""按实验开关执行共享 OCR det 模型调用。"""
|
||||
return _run_with_inference_lock(
|
||||
PIPELINE_OCR_DET_INFERENCE_LOCK, inference_callable, *args, **kwargs
|
||||
)
|
||||
|
||||
|
||||
def run_ocr_rec_inference(inference_callable, *args, **kwargs):
|
||||
"""按实验开关执行共享 OCR rec 模型调用。"""
|
||||
return _run_with_inference_lock(
|
||||
PIPELINE_OCR_REC_INFERENCE_LOCK, inference_callable, *args, **kwargs
|
||||
)
|
||||
|
||||
MFR_MODEL = os.getenv('MINERU_FORMULA_CH_SUPPORT', 'False')
|
||||
if MFR_MODEL.lower() in ['true', '1', 'yes']:
|
||||
|
||||
@@ -1,24 +1,24 @@
|
||||
# Copyright (c) Opendatalab. All rights reserved.
|
||||
import copy
|
||||
import os
|
||||
|
||||
from tqdm import tqdm
|
||||
|
||||
from mineru.backend.utils.html_image_utils import replace_inline_table_images
|
||||
from mineru.backend.utils.runtime_utils import cross_page_table_merge
|
||||
from mineru.utils.config_reader import get_device
|
||||
from mineru.backend.pipeline.model_init import AtomModelSingleton
|
||||
from mineru.backend.pipeline.model_init import (
|
||||
AtomModelSingleton,
|
||||
run_ocr_rec_inference,
|
||||
)
|
||||
from mineru.backend.pipeline.para_split import para_split
|
||||
from mineru.utils.char_utils import full_to_half
|
||||
from mineru.utils.cut_image import cut_image_and_table
|
||||
from mineru.utils.enum_class import ContentType, BlockType
|
||||
from mineru.utils.title_level_postprocess import apply_title_leveling_to_pdf_info
|
||||
from mineru.utils.model_utils import clean_memory
|
||||
from mineru.backend.pipeline.pipeline_magic_model import MagicModel
|
||||
from mineru.utils.ocr_utils import OcrConfidence, rotate_vertical_crop_if_needed
|
||||
from mineru.version import __version__
|
||||
from mineru.utils.hash_utils import bytes_md5
|
||||
from mineru.utils.pdfium_guard import close_pdfium_document, pdfium_guard
|
||||
from mineru.utils.pdfium_guard import close_pdfium_child, close_pdfium_document, pdfium_guard
|
||||
|
||||
|
||||
def page_model_info_to_page_info(page_model_info, image_dict, page, image_writer, page_index, ocr_enable=False):
|
||||
@@ -77,20 +77,24 @@ def append_page_model_infos_to_middle_json(
|
||||
):
|
||||
for offset, (page_model_info, image_dict) in enumerate(zip(page_model_infos, images_list)):
|
||||
page_index = page_start_index + offset
|
||||
with pdfium_guard():
|
||||
page = pdf_doc[page_index]
|
||||
page_info = page_model_info_to_page_info(
|
||||
copy.deepcopy(page_model_info),
|
||||
image_dict,
|
||||
page,
|
||||
image_writer,
|
||||
page_index,
|
||||
ocr_enable=ocr_enable,
|
||||
)
|
||||
if page_info is None:
|
||||
page = None
|
||||
try:
|
||||
with pdfium_guard():
|
||||
page_w, page_h = map(int, pdf_doc[page_index].get_size())
|
||||
page_info = make_page_info_dict([], page_index, page_w, page_h, [])
|
||||
page = pdf_doc[page_index]
|
||||
page_info = page_model_info_to_page_info(
|
||||
copy.deepcopy(page_model_info),
|
||||
image_dict,
|
||||
page,
|
||||
image_writer,
|
||||
page_index,
|
||||
ocr_enable=ocr_enable,
|
||||
)
|
||||
if page_info is None:
|
||||
with pdfium_guard():
|
||||
page_w, page_h = map(int, page.get_size())
|
||||
page_info = make_page_info_dict([], page_index, page_w, page_h, [])
|
||||
finally:
|
||||
close_pdfium_child(page)
|
||||
middle_json["pdf_info"].append(page_info)
|
||||
if progress_bar is not None:
|
||||
progress_bar.update(1)
|
||||
@@ -231,7 +235,9 @@ def _apply_post_ocr(pdf_info_list, lang=None):
|
||||
det_db_box_thresh=0.3,
|
||||
lang=lang
|
||||
)
|
||||
ocr_res_list = ocr_model.ocr(img_crop_list, det=False, tqdm_enable=True)[0]
|
||||
ocr_res_list = run_ocr_rec_inference(
|
||||
ocr_model.ocr, img_crop_list, det=False, tqdm_enable=True
|
||||
)[0]
|
||||
assert len(ocr_res_list) == len(
|
||||
need_ocr_list), f'ocr_res_list: {len(ocr_res_list)}, need_ocr_list: {len(need_ocr_list)}'
|
||||
for index, span in enumerate(need_ocr_list):
|
||||
@@ -280,8 +286,6 @@ def finalize_middle_json(
|
||||
"""Apply document-level post processing once all page_info entries are ready."""
|
||||
apply_server_side_postprocess(pdf_info_list, lang=lang)
|
||||
finalize_middle_json_from_preproc(pdf_info_list)
|
||||
if os.getenv('MINERU_DONOT_CLEAN_MEM') is None and len(pdf_info_list) >= 10:
|
||||
clean_memory(get_device())
|
||||
|
||||
|
||||
def init_middle_json():
|
||||
|
||||
@@ -168,53 +168,56 @@ def doc_analyze_streaming(
|
||||
raise ValueError("pdf_bytes_list, image_writer_list, and lang_list must have the same length")
|
||||
|
||||
doc_contexts = []
|
||||
total_pages = 0
|
||||
for doc_index, (pdf_bytes, image_writer, lang) in enumerate(
|
||||
zip(pdf_bytes_list, image_writer_list, lang_list)
|
||||
):
|
||||
_ocr_enable = _get_ocr_enable(pdf_bytes, parse_method)
|
||||
pdf_doc = open_pdfium_document(pdfium.PdfDocument, pdf_bytes)
|
||||
page_count = get_pdfium_document_page_count(pdf_doc)
|
||||
total_pages += page_count
|
||||
doc_contexts.append(
|
||||
{
|
||||
'doc_index': doc_index,
|
||||
'pdf_bytes': pdf_bytes,
|
||||
'pdf_doc': pdf_doc,
|
||||
'page_count': page_count,
|
||||
'next_page_idx': 0,
|
||||
'middle_json': init_middle_json(),
|
||||
'model_list': [],
|
||||
'image_writer': image_writer,
|
||||
'lang': lang,
|
||||
'ocr_enable': _ocr_enable,
|
||||
'closed': False,
|
||||
}
|
||||
try:
|
||||
total_pages = 0
|
||||
for doc_index, (pdf_bytes, image_writer, lang) in enumerate(
|
||||
zip(pdf_bytes_list, image_writer_list, lang_list)
|
||||
):
|
||||
_ocr_enable = _get_ocr_enable(pdf_bytes, parse_method)
|
||||
pdf_doc = open_pdfium_document(pdfium.PdfDocument, pdf_bytes)
|
||||
try:
|
||||
page_count = get_pdfium_document_page_count(pdf_doc)
|
||||
context = {
|
||||
'doc_index': doc_index,
|
||||
'pdf_bytes': pdf_bytes,
|
||||
'pdf_doc': pdf_doc,
|
||||
'page_count': page_count,
|
||||
'next_page_idx': 0,
|
||||
'middle_json': init_middle_json(),
|
||||
'model_list': [],
|
||||
'image_writer': image_writer,
|
||||
'lang': lang,
|
||||
'ocr_enable': _ocr_enable,
|
||||
'closed': False,
|
||||
}
|
||||
except Exception:
|
||||
close_pdfium_document(pdf_doc)
|
||||
raise
|
||||
total_pages += page_count
|
||||
doc_contexts.append(context)
|
||||
|
||||
if total_pages == 0:
|
||||
_emit_zero_page_contexts(
|
||||
doc_contexts,
|
||||
on_doc_ready,
|
||||
client_side_output_generation=client_side_output_generation,
|
||||
)
|
||||
return
|
||||
|
||||
window_size = get_processing_window_size(default=64)
|
||||
total_batches = (total_pages + window_size - 1) // window_size
|
||||
logger.info(
|
||||
f'Pipeline processing-window multi-file run. doc_count={len(doc_contexts)}, '
|
||||
f'total_pages={total_pages}, window_size={window_size}, total_batches={total_batches}'
|
||||
)
|
||||
|
||||
if total_pages == 0:
|
||||
_emit_zero_page_contexts(
|
||||
doc_contexts,
|
||||
on_doc_ready,
|
||||
client_side_output_generation=client_side_output_generation,
|
||||
)
|
||||
return
|
||||
|
||||
window_size = get_processing_window_size(default=64)
|
||||
total_batches = (total_pages + window_size - 1) // window_size
|
||||
logger.info(
|
||||
f'Pipeline processing-window multi-file run. doc_count={len(doc_contexts)}, '
|
||||
f'total_pages={total_pages}, window_size={window_size}, total_batches={total_batches}'
|
||||
)
|
||||
|
||||
_emit_zero_page_contexts(
|
||||
doc_contexts,
|
||||
on_doc_ready,
|
||||
client_side_output_generation=client_side_output_generation,
|
||||
)
|
||||
processed_pages = 0
|
||||
infer_start = time.time()
|
||||
try:
|
||||
processed_pages = 0
|
||||
infer_start = time.time()
|
||||
progress_bar = None
|
||||
last_append_end_time = None
|
||||
try:
|
||||
@@ -264,45 +267,48 @@ def doc_analyze_streaming(
|
||||
f'batch_pages={len(batch_images)}, doc_slices={_format_doc_slices(batch_slices)}'
|
||||
)
|
||||
|
||||
batch_results = batch_image_analyze(
|
||||
batch_images,
|
||||
formula_enable=formula_enable,
|
||||
table_enable=table_enable,
|
||||
)
|
||||
if progress_bar is None:
|
||||
progress_bar = tqdm(total=total_pages, desc="Processing pages")
|
||||
else:
|
||||
exclude_progress_bar_idle_time(
|
||||
progress_bar,
|
||||
last_append_end_time,
|
||||
now=time.time(),
|
||||
try:
|
||||
batch_results = batch_image_analyze(
|
||||
batch_images,
|
||||
formula_enable=formula_enable,
|
||||
table_enable=table_enable,
|
||||
)
|
||||
|
||||
result_offset = 0
|
||||
for context, images_list, page_start, take_count in batch_payloads:
|
||||
result_slice = batch_results[result_offset: result_offset + take_count]
|
||||
append_batch_results_to_middle_json(
|
||||
context['middle_json'],
|
||||
result_slice,
|
||||
images_list,
|
||||
context['pdf_doc'],
|
||||
context['image_writer'],
|
||||
page_start_index=page_start,
|
||||
ocr_enable=context['ocr_enable'],
|
||||
model_list=context['model_list'],
|
||||
progress_bar=progress_bar,
|
||||
)
|
||||
result_offset += take_count
|
||||
_close_images(images_list)
|
||||
images_list.clear()
|
||||
|
||||
if context['next_page_idx'] >= context['page_count'] and not context['closed']:
|
||||
_finalize_processing_window_context(
|
||||
context,
|
||||
on_doc_ready,
|
||||
client_side_output_generation=client_side_output_generation,
|
||||
if progress_bar is None:
|
||||
progress_bar = tqdm(total=total_pages, desc="Processing pages")
|
||||
else:
|
||||
exclude_progress_bar_idle_time(
|
||||
progress_bar,
|
||||
last_append_end_time,
|
||||
now=time.time(),
|
||||
)
|
||||
|
||||
result_offset = 0
|
||||
for context, images_list, page_start, take_count in batch_payloads:
|
||||
result_slice = batch_results[result_offset: result_offset + take_count]
|
||||
append_batch_results_to_middle_json(
|
||||
context['middle_json'],
|
||||
result_slice,
|
||||
images_list,
|
||||
context['pdf_doc'],
|
||||
context['image_writer'],
|
||||
page_start_index=page_start,
|
||||
ocr_enable=context['ocr_enable'],
|
||||
model_list=context['model_list'],
|
||||
progress_bar=progress_bar,
|
||||
)
|
||||
result_offset += take_count
|
||||
|
||||
if context['next_page_idx'] >= context['page_count'] and not context['closed']:
|
||||
_finalize_processing_window_context(
|
||||
context,
|
||||
on_doc_ready,
|
||||
client_side_output_generation=client_side_output_generation,
|
||||
)
|
||||
finally:
|
||||
for _context, images_list, _page_start, _take_count in batch_payloads:
|
||||
_close_images(images_list)
|
||||
images_list.clear()
|
||||
|
||||
last_append_end_time = time.time()
|
||||
processed_pages += len(batch_images)
|
||||
finally:
|
||||
|
||||
@@ -16,7 +16,7 @@ from mineru.utils.cut_image import cut_image_and_table
|
||||
from mineru.utils.enum_class import ContentType
|
||||
from mineru.utils.hash_utils import bytes_md5
|
||||
from mineru.utils.title_level_postprocess import apply_title_leveling_to_pdf_info
|
||||
from mineru.utils.pdfium_guard import close_pdfium_document, pdfium_guard
|
||||
from mineru.utils.pdfium_guard import close_pdfium_child, close_pdfium_document, pdfium_guard
|
||||
from mineru.version import __version__
|
||||
|
||||
|
||||
@@ -91,9 +91,13 @@ def append_page_blocks_to_middle_json(
|
||||
):
|
||||
for offset, (page_blocks, image_dict) in enumerate(zip(model_output_blocks_list, images_list)):
|
||||
page_index = page_start_index + offset
|
||||
with pdfium_guard():
|
||||
page = pdf_doc[page_index]
|
||||
page_info = blocks_to_page_info(page_blocks, image_dict, page, image_writer, page_index)
|
||||
page = None
|
||||
try:
|
||||
with pdfium_guard():
|
||||
page = pdf_doc[page_index]
|
||||
page_info = blocks_to_page_info(page_blocks, image_dict, page, image_writer, page_index)
|
||||
finally:
|
||||
close_pdfium_child(page)
|
||||
middle_json["pdf_info"].append(page_info)
|
||||
if progress_bar is not None:
|
||||
progress_bar.update(1)
|
||||
|
||||
@@ -161,7 +161,7 @@ async def parse_request_form(
|
||||
Form(
|
||||
description=(
|
||||
"Defer final markdown/content-list generation to the client. "
|
||||
"When enabled, the server returns staged middle JSON and images."
|
||||
"When enabled, the server returns staged middle JSON, model output, and images."
|
||||
),
|
||||
),
|
||||
] = False,
|
||||
@@ -188,7 +188,7 @@ async def parse_request_form(
|
||||
if client_side_output_generation:
|
||||
return_md = False
|
||||
return_middle_json = True
|
||||
return_model_output = False
|
||||
return_model_output = True
|
||||
return_content_list = False
|
||||
return_images = True
|
||||
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
# Copyright (c) Opendatalab. All rights reserved.
|
||||
import asyncio
|
||||
import multiprocessing
|
||||
import os
|
||||
import sys
|
||||
import threading
|
||||
@@ -343,8 +344,12 @@ async def mark_task_completed(
|
||||
|
||||
def create_visualization_context() -> Optional[VisualizationContext]:
|
||||
try:
|
||||
spawn_context = multiprocessing.get_context("spawn")
|
||||
return VisualizationContext(
|
||||
executor=ProcessPoolExecutor(max_workers=1),
|
||||
executor=ProcessPoolExecutor(
|
||||
max_workers=1,
|
||||
mp_context=spawn_context,
|
||||
),
|
||||
futures=[],
|
||||
)
|
||||
except Exception as exc:
|
||||
@@ -626,9 +631,8 @@ def build_request_form_data(
|
||||
image_analysis: bool = True,
|
||||
client_side_output_generation: bool = False,
|
||||
) -> dict[str, str | list[str]]:
|
||||
# 开启客户端输出生成时,服务端只负责返回 middle json、图片和原文件。
|
||||
# 开启客户端输出生成时,只关闭客户端会重建的最终产物。
|
||||
return_md = not client_side_output_generation
|
||||
return_model_output = not client_side_output_generation
|
||||
return_content_list = not client_side_output_generation
|
||||
return _api_client.build_parse_request_form_data(
|
||||
lang_list=[lang],
|
||||
@@ -642,7 +646,7 @@ def build_request_form_data(
|
||||
end_page_id=end_page_id,
|
||||
return_md=return_md,
|
||||
return_middle_json=True,
|
||||
return_model_output=return_model_output,
|
||||
return_model_output=True,
|
||||
return_content_list=return_content_list,
|
||||
return_images=True,
|
||||
response_format_zip=True,
|
||||
@@ -1128,7 +1132,7 @@ async def run_orchestrated_cli(
|
||||
@click.option(
|
||||
"--client-side-output-generation",
|
||||
"client_side_output_generation",
|
||||
is_flag=True,
|
||||
type=bool,
|
||||
default=False,
|
||||
help=(
|
||||
"Generate markdown and content lists locally from server-returned "
|
||||
|
||||
+43
-2
@@ -23,7 +23,10 @@ from mineru.backend.vlm.vlm_analyze import aio_doc_analyze as aio_vlm_doc_analyz
|
||||
from mineru.backend.office.pptx_analyze import office_pptx_analyze
|
||||
from mineru.backend.office.xlsx_analyze import office_xlsx_analyze
|
||||
from mineru.backend.office.docx_analyze import office_docx_analyze
|
||||
from mineru.utils.pdfium_guard import rewrite_pdf_bytes_with_pdfium
|
||||
from mineru.utils.pdfium_guard import (
|
||||
get_loadable_pdfium_page_indices,
|
||||
rewrite_pdf_bytes_with_pdfium,
|
||||
)
|
||||
|
||||
os.environ["TORCH_CUDNN_V8_API_DISABLED"] = "1"
|
||||
if os.getenv("MINERU_LMDEPLOY_DEVICE", "") == "maca":
|
||||
@@ -190,12 +193,50 @@ def convert_pdf_bytes_to_bytes(pdf_bytes, start_page_id=0, end_page_id=None):
|
||||
)
|
||||
if rebuilt_pdf_bytes:
|
||||
return rebuilt_pdf_bytes
|
||||
logger.warning("PDFium rewrite returned empty bytes, using original PDF bytes.")
|
||||
logger.warning(
|
||||
"PDFium rewrite returned empty bytes, trying to skip broken pages."
|
||||
)
|
||||
except Exception as fallback_error:
|
||||
logger.warning(
|
||||
f"Error in converting PDF bytes with pdfium: {fallback_error}, "
|
||||
"trying to skip broken pages."
|
||||
)
|
||||
|
||||
try:
|
||||
loadable_page_indices, broken_page_indices = get_loadable_pdfium_page_indices(
|
||||
pdf_bytes,
|
||||
start_page_id=start_page_id,
|
||||
end_page_id=end_page_id,
|
||||
)
|
||||
if broken_page_indices:
|
||||
skipped_pages = [page_index + 1 for page_index in broken_page_indices]
|
||||
logger.warning(
|
||||
f"Skipped broken PDF pages during PDFium rewrite: {skipped_pages}"
|
||||
)
|
||||
if not loadable_page_indices:
|
||||
logger.warning(
|
||||
"PDFium skip-broken-page rewrite found no loadable pages, "
|
||||
"using original PDF bytes."
|
||||
)
|
||||
return pdf_bytes
|
||||
|
||||
rebuilt_pdf_bytes = rewrite_pdf_bytes_with_pdfium(
|
||||
pdf_bytes,
|
||||
start_page_id=start_page_id,
|
||||
end_page_id=end_page_id,
|
||||
page_indices=loadable_page_indices,
|
||||
)
|
||||
if rebuilt_pdf_bytes:
|
||||
return rebuilt_pdf_bytes
|
||||
logger.warning(
|
||||
"PDFium skip-broken-page rewrite returned empty bytes, "
|
||||
"using original PDF bytes."
|
||||
)
|
||||
except Exception as fallback_error:
|
||||
logger.warning(
|
||||
"Error in converting PDF bytes with skip-broken-page fallback: "
|
||||
f"{fallback_error}, using original PDF bytes."
|
||||
)
|
||||
return pdf_bytes
|
||||
|
||||
|
||||
|
||||
@@ -997,7 +997,7 @@ async def _run_to_markdown_job(
|
||||
end_page_id=end_pages - 1,
|
||||
return_md=not use_client_side_output_generation,
|
||||
return_middle_json=True,
|
||||
return_model_output=not use_client_side_output_generation,
|
||||
return_model_output=True,
|
||||
return_content_list=not use_client_side_output_generation,
|
||||
return_images=True,
|
||||
response_format_zip=True,
|
||||
|
||||
@@ -112,7 +112,7 @@
|
||||
</span>
|
||||
<!-- Code Link. -->
|
||||
<span class="link-block">
|
||||
<a href="https://huggingface.co/opendatalab/MinerU2.5-Pro-2604-1.2B" target="_blank" class="external-link button is-normal is-rounded is-dark" style="text-decoration: none; cursor: pointer">
|
||||
<a href="https://huggingface.co/opendatalab/MinerU2.5-Pro-2605-1.2B" target="_blank" class="external-link button is-normal is-rounded is-dark" style="text-decoration: none; cursor: pointer">
|
||||
<span class="icon" style="margin-right: 4px">
|
||||
<i class="fas fa-cube" style="color: white; margin-right: 4px"></i>
|
||||
</span>
|
||||
|
||||
+239
-64
@@ -1,5 +1,6 @@
|
||||
# Copyright (c) Opendatalab. All rights reserved.
|
||||
import re
|
||||
from ctypes import byref, c_int, create_string_buffer
|
||||
from io import BytesIO
|
||||
|
||||
import pypdfium2 as pdfium
|
||||
@@ -7,6 +8,7 @@ import pypdfium2.raw as pdfium_c
|
||||
from loguru import logger
|
||||
from pypdf import PdfReader
|
||||
from mineru.utils.pdfium_guard import (
|
||||
close_pdfium_child,
|
||||
close_pdfium_document,
|
||||
open_pdfium_document,
|
||||
pdfium_guard,
|
||||
@@ -18,6 +20,8 @@ HIGH_IMAGE_COVERAGE_THRESHOLD = 0.8
|
||||
TEXT_QUALITY_MIN_CHARS = 300
|
||||
TEXT_QUALITY_BAD_THRESHOLD = 0.03
|
||||
UNICODE_MAP_ERROR_RATIO_THRESHOLD = 0.04
|
||||
CID_FONT_USAGE_RATIO_THRESHOLD = 0.01
|
||||
CID_FONT_USAGE_COUNT_THRESHOLD = 30
|
||||
MAX_PAGE_ASPECT_RATIO = 10.0
|
||||
SUSPICIOUS_CJK_72XX_START = 0x7280
|
||||
SUSPICIOUS_CJK_72XX_END = 0x72DF
|
||||
@@ -103,7 +107,20 @@ def classify(pdf_bytes):
|
||||
)
|
||||
return "ocr"
|
||||
|
||||
if detect_cid_font_signal_pypdf(pdf_bytes, page_indices):
|
||||
cid_font_signal = get_cid_font_signal_pypdf(pdf_bytes, page_indices)
|
||||
cid_font_usage_signal = _get_cid_font_usage_signal_from_samples(
|
||||
text_samples,
|
||||
cid_font_signal,
|
||||
)
|
||||
if cid_font_usage_signal["triggered"]:
|
||||
logger.debug(
|
||||
"Classify PDF as OCR due to high CID font usage without ToUnicode: "
|
||||
f"page={cid_font_usage_signal['page_index'] + 1}, "
|
||||
f"fonts={cid_font_usage_signal['font_names']}, "
|
||||
f"chars={cid_font_usage_signal['cid_font_char_count']}, "
|
||||
f"total={cid_font_usage_signal['total_chars']}, "
|
||||
f"ratio={cid_font_usage_signal['cid_font_usage_ratio']:.4f}"
|
||||
)
|
||||
return "ocr"
|
||||
|
||||
text_quality_signal = _get_text_quality_signal_from_samples(text_samples)
|
||||
@@ -196,35 +213,88 @@ def get_extreme_aspect_ratio_page_pdfium(
|
||||
page_indices,
|
||||
max_page_aspect_ratio: float = MAX_PAGE_ASPECT_RATIO,
|
||||
):
|
||||
for page_index in page_indices:
|
||||
page = pdf_doc[page_index]
|
||||
page_width, page_height = page.get_size()
|
||||
if page_width <= 0 or page_height <= 0:
|
||||
continue
|
||||
with pdfium_guard():
|
||||
for page_index in page_indices:
|
||||
page = None
|
||||
try:
|
||||
page = pdf_doc[page_index]
|
||||
page_width, page_height = page.get_size()
|
||||
if page_width <= 0 or page_height <= 0:
|
||||
continue
|
||||
|
||||
aspect_ratio = max(page_width / page_height, page_height / page_width)
|
||||
if aspect_ratio > max_page_aspect_ratio:
|
||||
return page_index, aspect_ratio
|
||||
aspect_ratio = max(page_width / page_height, page_height / page_width)
|
||||
if aspect_ratio > max_page_aspect_ratio:
|
||||
return page_index, aspect_ratio
|
||||
finally:
|
||||
close_pdfium_child(page)
|
||||
|
||||
return None, None
|
||||
|
||||
|
||||
def _collect_pdfium_text_samples(pdf_doc, page_indices):
|
||||
"""一次性收集抽样页的 textpage 和文本,避免分类链路重复读取 PDFium 文本。"""
|
||||
text_samples = []
|
||||
|
||||
for page_index in page_indices:
|
||||
page = pdf_doc[page_index]
|
||||
def _collect_pdfium_text_sample_from_page(page_index, page):
|
||||
"""从单页 PDFium 对象提取纯 Python 文本统计,并在调用方释放子对象。"""
|
||||
text_page = None
|
||||
try:
|
||||
text_page = page.get_textpage()
|
||||
text = text_page.get_text_bounded()
|
||||
text_samples.append(
|
||||
{
|
||||
"page_index": page_index,
|
||||
"text_page": text_page,
|
||||
"text": text,
|
||||
"cleaned_text": re.sub(r"\s+", "", text),
|
||||
}
|
||||
)
|
||||
char_count = text_page.count_chars()
|
||||
null_char_count = 0
|
||||
replacement_char_count = 0
|
||||
control_char_count = 0
|
||||
private_use_char_count = 0
|
||||
unicode_map_error_count = 0
|
||||
font_name_counts = {}
|
||||
|
||||
for char_index in range(char_count):
|
||||
unicode_code = pdfium_c.FPDFText_GetUnicode(text_page, char_index)
|
||||
if unicode_code == 0:
|
||||
null_char_count += 1
|
||||
elif unicode_code == 0xFFFD:
|
||||
replacement_char_count += 1
|
||||
elif _is_disallowed_control_unicode(unicode_code):
|
||||
control_char_count += 1
|
||||
elif _PRIVATE_USE_AREA_START <= unicode_code <= _PRIVATE_USE_AREA_END:
|
||||
private_use_char_count += 1
|
||||
|
||||
if pdfium_c.FPDFText_HasUnicodeMapError(text_page, char_index):
|
||||
unicode_map_error_count += 1
|
||||
|
||||
font_name = _normalize_pdf_font_name(
|
||||
_get_pdfium_char_font_name(text_page, char_index)
|
||||
)
|
||||
if font_name:
|
||||
font_name_counts[font_name] = font_name_counts.get(font_name, 0) + 1
|
||||
|
||||
return {
|
||||
"page_index": page_index,
|
||||
"text": text,
|
||||
"cleaned_text": re.sub(r"\s+", "", text),
|
||||
"char_count": char_count,
|
||||
"null_char_count": null_char_count,
|
||||
"replacement_char_count": replacement_char_count,
|
||||
"control_char_count": control_char_count,
|
||||
"private_use_char_count": private_use_char_count,
|
||||
"unicode_map_error_count": unicode_map_error_count,
|
||||
"font_name_counts": font_name_counts,
|
||||
}
|
||||
finally:
|
||||
close_pdfium_child(text_page)
|
||||
|
||||
|
||||
def _collect_pdfium_text_samples(pdf_doc, page_indices):
|
||||
"""一次性收集抽样页文本统计,返回纯 Python 数据,避免缓存 PDFium 子对象。"""
|
||||
text_samples = []
|
||||
|
||||
with pdfium_guard():
|
||||
for page_index in page_indices:
|
||||
page = None
|
||||
try:
|
||||
page = pdf_doc[page_index]
|
||||
text_samples.append(
|
||||
_collect_pdfium_text_sample_from_page(page_index, page)
|
||||
)
|
||||
finally:
|
||||
close_pdfium_child(page)
|
||||
|
||||
return text_samples
|
||||
|
||||
@@ -247,7 +317,7 @@ def get_avg_cleaned_chars_per_page_pdfium(pdf_doc, page_indices):
|
||||
|
||||
|
||||
def _get_text_quality_signal_from_samples(text_samples):
|
||||
"""基于已缓存的 PDFium textpage 统计异常字符质量信号。"""
|
||||
"""基于已缓存的抽样页字符计数统计异常字符质量信号。"""
|
||||
total_chars = 0
|
||||
null_char_count = 0
|
||||
replacement_char_count = 0
|
||||
@@ -255,20 +325,11 @@ def _get_text_quality_signal_from_samples(text_samples):
|
||||
private_use_char_count = 0
|
||||
|
||||
for text_sample in text_samples:
|
||||
text_page = text_sample["text_page"]
|
||||
char_count = text_page.count_chars()
|
||||
total_chars += char_count
|
||||
|
||||
for char_index in range(char_count):
|
||||
unicode_code = pdfium_c.FPDFText_GetUnicode(text_page, char_index)
|
||||
if unicode_code == 0:
|
||||
null_char_count += 1
|
||||
elif unicode_code == 0xFFFD:
|
||||
replacement_char_count += 1
|
||||
elif _is_disallowed_control_unicode(unicode_code):
|
||||
control_char_count += 1
|
||||
elif _PRIVATE_USE_AREA_START <= unicode_code <= _PRIVATE_USE_AREA_END:
|
||||
private_use_char_count += 1
|
||||
total_chars += text_sample["char_count"]
|
||||
null_char_count += text_sample["null_char_count"]
|
||||
replacement_char_count += text_sample["replacement_char_count"]
|
||||
control_char_count += text_sample["control_char_count"]
|
||||
private_use_char_count += text_sample["private_use_char_count"]
|
||||
|
||||
abnormal_chars = (
|
||||
null_char_count
|
||||
@@ -302,13 +363,8 @@ def _get_unicode_map_error_signal_from_samples(text_samples):
|
||||
unicode_map_error_count = 0
|
||||
|
||||
for text_sample in text_samples:
|
||||
text_page = text_sample["text_page"]
|
||||
char_count = text_page.count_chars()
|
||||
total_chars += char_count
|
||||
|
||||
for char_index in range(char_count):
|
||||
if pdfium_c.FPDFText_HasUnicodeMapError(text_page, char_index):
|
||||
unicode_map_error_count += 1
|
||||
total_chars += text_sample["char_count"]
|
||||
unicode_map_error_count += text_sample["unicode_map_error_count"]
|
||||
|
||||
unicode_map_error_ratio = 0.0
|
||||
if total_chars > 0:
|
||||
@@ -321,6 +377,102 @@ def _get_unicode_map_error_signal_from_samples(text_samples):
|
||||
}
|
||||
|
||||
|
||||
def _normalize_pdf_font_name(font_name) -> str:
|
||||
"""规范化 PDF 字体名,统一 pypdf 的 NameObject 和 PDFium 返回值格式。"""
|
||||
if font_name is None:
|
||||
return ""
|
||||
return str(font_name).strip().lstrip("/")
|
||||
|
||||
|
||||
def _get_pdfium_char_font_name(text_page, char_index: int) -> str:
|
||||
"""读取 PDFium 字符级字体名,用于统计可疑 CID 字体的实际使用比例。"""
|
||||
flags = c_int()
|
||||
buffer_length = pdfium_c.FPDFText_GetFontInfo(
|
||||
text_page,
|
||||
char_index,
|
||||
None,
|
||||
0,
|
||||
byref(flags),
|
||||
)
|
||||
if buffer_length <= 0:
|
||||
return ""
|
||||
|
||||
font_name_buffer = create_string_buffer(buffer_length)
|
||||
actual_length = pdfium_c.FPDFText_GetFontInfo(
|
||||
text_page,
|
||||
char_index,
|
||||
font_name_buffer,
|
||||
buffer_length,
|
||||
byref(flags),
|
||||
)
|
||||
if actual_length <= 0:
|
||||
return ""
|
||||
|
||||
return font_name_buffer.value.decode("utf-8", errors="ignore")
|
||||
|
||||
|
||||
def _get_cid_font_usage_signal_from_samples(text_samples, cid_font_signal):
|
||||
"""基于 PDFium 字符级字体名统计可疑 CID 字体在抽样页中的真实使用比例。"""
|
||||
best_signal = {
|
||||
"triggered": False,
|
||||
"page_index": None,
|
||||
"font_names": [],
|
||||
"cid_font_char_count": 0,
|
||||
"total_chars": 0,
|
||||
"cid_font_usage_ratio": 0.0,
|
||||
}
|
||||
if not cid_font_signal or not cid_font_signal.get("triggered"):
|
||||
return best_signal
|
||||
|
||||
page_fonts = cid_font_signal.get("page_fonts") or {}
|
||||
for text_sample in text_samples:
|
||||
page_index = text_sample.get("page_index")
|
||||
cid_font_names = {
|
||||
_normalize_pdf_font_name(font_name)
|
||||
for font_name in page_fonts.get(page_index, set())
|
||||
}
|
||||
cid_font_names.discard("")
|
||||
if not cid_font_names:
|
||||
continue
|
||||
|
||||
total_chars = text_sample["char_count"]
|
||||
if total_chars <= 0:
|
||||
continue
|
||||
|
||||
font_name_counts = text_sample.get("font_name_counts") or {}
|
||||
matched_font_names = cid_font_names.intersection(font_name_counts)
|
||||
cid_font_char_count = sum(
|
||||
font_name_counts[font_name] for font_name in matched_font_names
|
||||
)
|
||||
|
||||
cid_font_usage_ratio = cid_font_char_count / total_chars
|
||||
signal = {
|
||||
"triggered": False,
|
||||
"page_index": page_index,
|
||||
"font_names": sorted(matched_font_names),
|
||||
"cid_font_char_count": cid_font_char_count,
|
||||
"total_chars": total_chars,
|
||||
"cid_font_usage_ratio": cid_font_usage_ratio,
|
||||
}
|
||||
if (
|
||||
cid_font_char_count >= CID_FONT_USAGE_COUNT_THRESHOLD
|
||||
and cid_font_usage_ratio >= CID_FONT_USAGE_RATIO_THRESHOLD
|
||||
):
|
||||
signal["triggered"] = True
|
||||
return signal
|
||||
|
||||
if (
|
||||
signal["cid_font_usage_ratio"],
|
||||
signal["cid_font_char_count"],
|
||||
) > (
|
||||
best_signal["cid_font_usage_ratio"],
|
||||
best_signal["cid_font_char_count"],
|
||||
):
|
||||
best_signal = signal
|
||||
|
||||
return best_signal
|
||||
|
||||
|
||||
def _get_u72xx_text_signal_from_samples(text_samples):
|
||||
"""基于已缓存的抽样页文本统计扣除常用字后的 U+7280-U+72DF 字符占比。"""
|
||||
cjk_chars = 0
|
||||
@@ -435,8 +587,10 @@ def _get_sampled_ascii_punct_signal_from_samples(text_samples):
|
||||
return best_signal
|
||||
|
||||
|
||||
def detect_cid_font_signal_pypdf(pdf_bytes, page_indices):
|
||||
def get_cid_font_signal_pypdf(pdf_bytes, page_indices):
|
||||
"""收集抽样页中无 ToUnicode 的 Identity CID 字体资源,供后续按实际字符使用量判定。"""
|
||||
reader = PdfReader(BytesIO(pdf_bytes))
|
||||
page_fonts = {}
|
||||
|
||||
for page_index in page_indices:
|
||||
page = reader.pages[page_index]
|
||||
@@ -448,7 +602,7 @@ def detect_cid_font_signal_pypdf(pdf_bytes, page_indices):
|
||||
if not fonts:
|
||||
continue
|
||||
|
||||
for _, font_ref in fonts.items():
|
||||
for font_key, font_ref in fonts.items():
|
||||
font = _resolve_pdf_object(font_ref)
|
||||
if not font:
|
||||
continue
|
||||
@@ -464,9 +618,20 @@ def detect_cid_font_signal_pypdf(pdf_bytes, page_indices):
|
||||
and has_descendant_fonts
|
||||
and not has_to_unicode
|
||||
):
|
||||
return True
|
||||
font_name = font.get("/BaseFont") or font_key
|
||||
page_fonts.setdefault(page_index, set()).add(
|
||||
_normalize_pdf_font_name(font_name)
|
||||
)
|
||||
|
||||
return False
|
||||
return {
|
||||
"triggered": bool(page_fonts),
|
||||
"page_fonts": page_fonts,
|
||||
}
|
||||
|
||||
|
||||
def detect_cid_font_signal_pypdf(pdf_bytes, page_indices):
|
||||
"""兼容旧接口:只返回是否存在无 ToUnicode 的 Identity CID 字体资源。"""
|
||||
return get_cid_font_signal_pypdf(pdf_bytes, page_indices)["triggered"]
|
||||
|
||||
|
||||
def _resolve_pdf_object(obj):
|
||||
@@ -478,23 +643,33 @@ def _resolve_pdf_object(obj):
|
||||
def get_high_image_coverage_ratio_pdfium(pdf_doc, page_indices):
|
||||
high_image_coverage_pages = 0
|
||||
|
||||
for page_index in page_indices:
|
||||
page = pdf_doc[page_index]
|
||||
page_bbox = page.get_bbox()
|
||||
page_area = abs(
|
||||
(page_bbox[2] - page_bbox[0]) * (page_bbox[3] - page_bbox[1])
|
||||
)
|
||||
image_area = 0.0
|
||||
with pdfium_guard():
|
||||
for page_index in page_indices:
|
||||
page = None
|
||||
try:
|
||||
page = pdf_doc[page_index]
|
||||
page_bbox = page.get_bbox()
|
||||
page_area = abs(
|
||||
(page_bbox[2] - page_bbox[0]) * (page_bbox[3] - page_bbox[1])
|
||||
)
|
||||
image_area = 0.0
|
||||
|
||||
for page_object in page.get_objects(
|
||||
filter=[pdfium_c.FPDF_PAGEOBJ_IMAGE], max_depth=3
|
||||
):
|
||||
left, bottom, right, top = page_object.get_pos()
|
||||
image_area += max(0.0, right - left) * max(0.0, top - bottom)
|
||||
for page_object in page.get_objects(
|
||||
filter=[pdfium_c.FPDF_PAGEOBJ_IMAGE], max_depth=3
|
||||
):
|
||||
try:
|
||||
left, bottom, right, top = page_object.get_pos()
|
||||
image_area += max(0.0, right - left) * max(0.0, top - bottom)
|
||||
finally:
|
||||
close_pdfium_child(page_object)
|
||||
|
||||
coverage_ratio = min(image_area / page_area, 1.0) if page_area > 0 else 0.0
|
||||
if coverage_ratio >= HIGH_IMAGE_COVERAGE_THRESHOLD:
|
||||
high_image_coverage_pages += 1
|
||||
coverage_ratio = (
|
||||
min(image_area / page_area, 1.0) if page_area > 0 else 0.0
|
||||
)
|
||||
if coverage_ratio >= HIGH_IMAGE_COVERAGE_THRESHOLD:
|
||||
high_image_coverage_pages += 1
|
||||
finally:
|
||||
close_pdfium_child(page)
|
||||
|
||||
if not page_indices:
|
||||
return 0.0
|
||||
|
||||
@@ -21,6 +21,7 @@ from mineru.utils.enum_class import ImageType
|
||||
from mineru.utils.hash_utils import str_sha256
|
||||
from mineru.utils.pdf_page_id import get_end_page_id
|
||||
from mineru.utils.pdfium_guard import (
|
||||
close_pdfium_child,
|
||||
close_pdfium_document,
|
||||
get_pdfium_document_page_count,
|
||||
open_pdfium_document,
|
||||
@@ -35,11 +36,15 @@ DEFAULT_PDF_IMAGE_DPI = 200
|
||||
# DEFAULT_PDF_IMAGE_DPI = 144
|
||||
MAX_PDF_RENDER_PROCESSES = 3
|
||||
MIN_PAGES_PER_RENDER_PROCESS = 30
|
||||
PDF_RENDER_PROCESS_SPAWN_DELAY_SECONDS = 0.1
|
||||
PDF_RENDER_TERMINATE_GRACE_PERIOD_SECONDS = 0.1
|
||||
PDF_RENDER_KILL_JOIN_TIMEOUT_SECONDS = 0.1
|
||||
|
||||
_pdf_render_executor: ProcessPoolExecutor | None = None
|
||||
_pdf_render_executor_lock = threading.Lock()
|
||||
_pdf_render_spawn_submit_lock = threading.Lock()
|
||||
_pdf_render_spawn_submit_executor_id: int | None = None
|
||||
_pdf_render_spawn_submit_count = 0
|
||||
|
||||
|
||||
def pdf_page_to_image(
|
||||
@@ -62,7 +67,10 @@ def pdf_page_to_image(
|
||||
"scale": scale,
|
||||
}
|
||||
if image_type == ImageType.BASE64:
|
||||
image_dict["img_base64"] = image_to_b64str(pil_img)
|
||||
try:
|
||||
image_dict["img_base64"] = image_to_b64str(pil_img)
|
||||
finally:
|
||||
pil_img.close()
|
||||
else:
|
||||
image_dict["img_pil"] = pil_img
|
||||
|
||||
@@ -78,6 +86,18 @@ def _load_images_from_pdf_worker(
|
||||
)
|
||||
|
||||
|
||||
def _close_image_dicts(images_list) -> None:
|
||||
"""关闭 image dict 中的 PIL 图片,供异常清理路径释放已生成的图像资源。"""
|
||||
for image_dict in images_list or []:
|
||||
pil_img = image_dict.get("img_pil")
|
||||
if pil_img is None:
|
||||
continue
|
||||
try:
|
||||
pil_img.close()
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
|
||||
def _calculate_render_process_count(total_pages: int, threads: int, cpu_count=None) -> int:
|
||||
requested_threads = max(1, threads)
|
||||
available_cpus = max(1, cpu_count if cpu_count is not None else (os.cpu_count() or 1))
|
||||
@@ -137,9 +157,10 @@ def _create_pdf_render_executor(max_workers: int) -> ProcessPoolExecutor:
|
||||
return ProcessPoolExecutor(max_workers=max_workers)
|
||||
|
||||
start_method = multiprocessing.get_start_method()
|
||||
if start_method == "fork":
|
||||
if start_method != "spawn":
|
||||
logger.debug(
|
||||
"PDF image rendering switches multiprocessing start method from fork to spawn"
|
||||
"PDF image rendering switches multiprocessing start method "
|
||||
f"from {start_method} to spawn"
|
||||
)
|
||||
return ProcessPoolExecutor(
|
||||
max_workers=max_workers,
|
||||
@@ -149,6 +170,44 @@ def _create_pdf_render_executor(max_workers: int) -> ProcessPoolExecutor:
|
||||
return ProcessPoolExecutor(max_workers=max_workers)
|
||||
|
||||
|
||||
def _is_pdf_render_pool_still_spawning_workers(executor: ProcessPoolExecutor) -> bool:
|
||||
"""判断渲染进程池是否还可能因为 submit 而继续创建新的 worker。"""
|
||||
max_workers = getattr(executor, "_max_workers", None)
|
||||
if max_workers is None or max_workers <= 1:
|
||||
return False
|
||||
|
||||
processes = getattr(executor, "_processes", None)
|
||||
process_count = 0 if processes is None else len(processes)
|
||||
return process_count < max_workers
|
||||
|
||||
|
||||
def _submit_pdf_render_task(
|
||||
executor: ProcessPoolExecutor,
|
||||
fn,
|
||||
*args,
|
||||
**kwargs,
|
||||
):
|
||||
"""提交 PDF 渲染任务;冷启动补 worker 时串行 submit 并错开 100ms。"""
|
||||
global _pdf_render_spawn_submit_executor_id, _pdf_render_spawn_submit_count
|
||||
|
||||
with _pdf_render_spawn_submit_lock:
|
||||
should_throttle = _is_pdf_render_pool_still_spawning_workers(executor)
|
||||
if should_throttle:
|
||||
executor_id = id(executor)
|
||||
if _pdf_render_spawn_submit_executor_id != executor_id:
|
||||
_pdf_render_spawn_submit_executor_id = executor_id
|
||||
_pdf_render_spawn_submit_count = 0
|
||||
elif _pdf_render_spawn_submit_count > 0:
|
||||
time.sleep(PDF_RENDER_PROCESS_SPAWN_DELAY_SECONDS)
|
||||
|
||||
future = executor.submit(fn, *args, **kwargs)
|
||||
|
||||
if should_throttle:
|
||||
_pdf_render_spawn_submit_count += 1
|
||||
|
||||
return future
|
||||
|
||||
|
||||
def _get_pdf_render_executor() -> ProcessPoolExecutor:
|
||||
global _pdf_render_executor
|
||||
|
||||
@@ -177,8 +236,14 @@ def _recycle_pdf_render_executor(
|
||||
_pdf_render_executor = None
|
||||
|
||||
if terminate_processes:
|
||||
_terminate_executor_processes(executor)
|
||||
executor.shutdown(wait=False, cancel_futures=True)
|
||||
try:
|
||||
_terminate_executor_processes(executor)
|
||||
except Exception as exc:
|
||||
logger.warning(f"Failed to terminate PDF render executor processes: {exc}")
|
||||
try:
|
||||
executor.shutdown(wait=False, cancel_futures=True)
|
||||
except Exception as exc:
|
||||
logger.warning(f"Failed to shutdown PDF render executor: {exc}")
|
||||
|
||||
|
||||
def shutdown_pdf_render_executor() -> None:
|
||||
@@ -228,11 +293,13 @@ def _load_images_from_pdf_bytes_range(
|
||||
|
||||
executor = _get_pdf_render_executor()
|
||||
recycle_executor = False
|
||||
collected_image_lists = []
|
||||
try:
|
||||
futures = []
|
||||
future_to_range = {}
|
||||
for range_start, range_end in page_ranges:
|
||||
future = executor.submit(
|
||||
future = _submit_pdf_render_task(
|
||||
executor,
|
||||
_load_images_from_pdf_worker,
|
||||
pdf_bytes,
|
||||
dpi,
|
||||
@@ -255,6 +322,7 @@ def _load_images_from_pdf_bytes_range(
|
||||
for future in futures:
|
||||
range_start = future_to_range[future]
|
||||
images_list = future.result()
|
||||
collected_image_lists.append(images_list)
|
||||
all_results.append((range_start, images_list))
|
||||
|
||||
all_results.sort(key=lambda x: x[0])
|
||||
@@ -262,10 +330,15 @@ def _load_images_from_pdf_bytes_range(
|
||||
for _, imgs in all_results:
|
||||
images_list.extend(imgs)
|
||||
|
||||
collected_image_lists.clear()
|
||||
return images_list
|
||||
except BrokenProcessPool:
|
||||
recycle_executor = True
|
||||
raise
|
||||
except Exception:
|
||||
for images_list in collected_image_lists:
|
||||
_close_image_dicts(images_list)
|
||||
raise
|
||||
finally:
|
||||
if recycle_executor:
|
||||
logger.warning("Recycling persistent PDF render executor after render failure")
|
||||
@@ -298,7 +371,9 @@ async def aio_load_images_from_pdf_bytes_range(
|
||||
|
||||
def _terminate_executor_processes(executor):
|
||||
"""强制终止 ProcessPoolExecutor 中的所有子进程"""
|
||||
processes = list(getattr(executor, "_processes", {}).values())
|
||||
# executor.shutdown() 后 _processes 会被置空,重复回收时直接视为无进程。
|
||||
process_map = getattr(executor, "_processes", None) or {}
|
||||
processes = list(process_map.values())
|
||||
if not processes:
|
||||
return
|
||||
|
||||
@@ -358,9 +433,13 @@ def load_images_from_pdf_core(
|
||||
|
||||
for index in range(start_page_id, end_page_id + 1):
|
||||
# logger.debug(f"Converting page {index}/{pdf_page_num} to image")
|
||||
page = pdf_doc[index]
|
||||
image_dict = pdf_page_to_image(page, dpi=dpi, image_type=image_type)
|
||||
images_list.append(image_dict)
|
||||
page = None
|
||||
try:
|
||||
page = pdf_doc[index]
|
||||
image_dict = pdf_page_to_image(page, dpi=dpi, image_type=image_type)
|
||||
images_list.append(image_dict)
|
||||
finally:
|
||||
close_pdfium_child(page)
|
||||
finally:
|
||||
close_pdfium_document(pdf_doc)
|
||||
|
||||
@@ -394,9 +473,13 @@ def load_images_from_pdf_doc(
|
||||
images_list = []
|
||||
with pdfium_guard():
|
||||
for index in range(start_page_id, normalized_end_page_id + 1):
|
||||
page = pdf_doc[index]
|
||||
image_dict = pdf_page_to_image(page, dpi=dpi, image_type=image_type)
|
||||
images_list.append(image_dict)
|
||||
page = None
|
||||
try:
|
||||
page = pdf_doc[index]
|
||||
image_dict = pdf_page_to_image(page, dpi=dpi, image_type=image_type)
|
||||
images_list.append(image_dict)
|
||||
finally:
|
||||
close_pdfium_child(page)
|
||||
|
||||
return images_list
|
||||
|
||||
|
||||
@@ -20,13 +20,16 @@ def page_to_image(
|
||||
if (long_side_length*scale) > max_width_or_height:
|
||||
scale = max_width_or_height / long_side_length
|
||||
|
||||
bitmap: PdfBitmap = page.render(scale=scale) # type: ignore
|
||||
|
||||
image = bitmap.to_pil()
|
||||
bitmap: PdfBitmap | None = None
|
||||
try:
|
||||
bitmap.close()
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to close bitmap: {e}")
|
||||
bitmap = page.render(scale=scale) # type: ignore
|
||||
image = bitmap.to_pil().copy()
|
||||
finally:
|
||||
if bitmap is not None:
|
||||
try:
|
||||
bitmap.close()
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to close bitmap: {e}")
|
||||
return image, scale
|
||||
|
||||
|
||||
|
||||
@@ -6,7 +6,7 @@ import pypdfium2 as pdfium
|
||||
from pdftext.pdf.chars import deduplicate_chars, get_chars
|
||||
from pdftext.pdf.pages import assign_scripts, get_blocks, get_lines, get_spans
|
||||
|
||||
from mineru.utils.pdfium_guard import pdfium_guard
|
||||
from mineru.utils.pdfium_guard import close_pdfium_child, pdfium_guard
|
||||
|
||||
|
||||
def get_page(
|
||||
@@ -39,25 +39,30 @@ def get_page_chars(
|
||||
page_char_count: int | None = None,
|
||||
) -> dict:
|
||||
"""轻量读取页面字符坐标,供只需要 char 级信息的路径复用。"""
|
||||
with pdfium_guard():
|
||||
if textpage is None:
|
||||
textpage = page.get_textpage()
|
||||
page_bbox: List[float] = page.get_bbox()
|
||||
page_width = math.ceil(abs(page_bbox[2] - page_bbox[0]))
|
||||
page_height = math.ceil(abs(page_bbox[1] - page_bbox[3]))
|
||||
owns_textpage = textpage is None
|
||||
try:
|
||||
with pdfium_guard():
|
||||
if textpage is None:
|
||||
textpage = page.get_textpage()
|
||||
page_bbox: List[float] = page.get_bbox()
|
||||
page_width = math.ceil(abs(page_bbox[2] - page_bbox[0]))
|
||||
page_height = math.ceil(abs(page_bbox[1] - page_bbox[3]))
|
||||
|
||||
page_rotation = 0
|
||||
try:
|
||||
page_rotation = page.get_rotation()
|
||||
except Exception:
|
||||
pass
|
||||
page_rotation = 0
|
||||
try:
|
||||
page_rotation = page.get_rotation()
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
if page_char_count is None:
|
||||
page_char_count = textpage.count_chars()
|
||||
if page_char_count is None:
|
||||
page_char_count = textpage.count_chars()
|
||||
|
||||
chars = deduplicate_chars(
|
||||
get_chars(textpage, page_bbox, page_rotation, quote_loosebox)
|
||||
)
|
||||
chars = deduplicate_chars(
|
||||
get_chars(textpage, page_bbox, page_rotation, quote_loosebox)
|
||||
)
|
||||
finally:
|
||||
if owns_textpage:
|
||||
close_pdfium_child(textpage)
|
||||
|
||||
return {
|
||||
"bbox": page_bbox,
|
||||
|
||||
@@ -4,6 +4,8 @@ from io import BytesIO
|
||||
from contextlib import contextmanager
|
||||
from typing import Any, Callable, Sequence, TypeVar
|
||||
|
||||
from loguru import logger
|
||||
|
||||
from mineru.utils.pdf_page_id import get_end_page_id
|
||||
|
||||
|
||||
@@ -39,6 +41,70 @@ def close_pdfium_document(pdf_doc) -> None:
|
||||
pdf_doc.close()
|
||||
|
||||
|
||||
def close_pdfium_child(pdfium_obj) -> None:
|
||||
"""显式关闭 PDFium 子对象,避免依赖 weakref/finalizer 延迟释放 native 资源。"""
|
||||
if pdfium_obj is None:
|
||||
return
|
||||
close = getattr(pdfium_obj, "close", None)
|
||||
if callable(close):
|
||||
with pdfium_guard():
|
||||
close()
|
||||
|
||||
|
||||
def close_pdfium_objects_safely(*pdfium_objs, owner: str = "pdfium cleanup") -> None:
|
||||
"""清理多个 PDFium 对象时逐个尝试关闭,避免前一个关闭失败阻断后续对象释放。"""
|
||||
for pdfium_obj in pdfium_objs:
|
||||
if pdfium_obj is None:
|
||||
continue
|
||||
try:
|
||||
close_pdfium_child(pdfium_obj)
|
||||
except Exception as exc:
|
||||
logger.warning(f"Failed to close PDFium object during {owner}: {exc}")
|
||||
|
||||
|
||||
def get_loadable_pdfium_page_indices(
|
||||
src_pdf_bytes: bytes,
|
||||
start_page_id: int = 0,
|
||||
end_page_id: int | None = None,
|
||||
) -> tuple[list[int], list[int]]:
|
||||
"""逐页探测 PDFium 可加载页面,返回可保留页和损坏页的 0-based 索引。"""
|
||||
import pypdfium2 as pdfium
|
||||
|
||||
loadable_page_indices = []
|
||||
broken_page_indices = []
|
||||
pdf_doc = None
|
||||
|
||||
try:
|
||||
with pdfium_guard():
|
||||
pdf_doc = pdfium.PdfDocument(src_pdf_bytes)
|
||||
total_page_count = len(pdf_doc)
|
||||
if total_page_count == 0:
|
||||
return [], []
|
||||
|
||||
normalized_start_page_id = max(0, start_page_id)
|
||||
normalized_end_page_id = get_end_page_id(end_page_id, total_page_count)
|
||||
if normalized_start_page_id > normalized_end_page_id:
|
||||
return [], []
|
||||
|
||||
for page_index in range(
|
||||
normalized_start_page_id,
|
||||
normalized_end_page_id + 1,
|
||||
):
|
||||
page = None
|
||||
try:
|
||||
page = pdf_doc[page_index]
|
||||
page.get_size()
|
||||
loadable_page_indices.append(page_index)
|
||||
except Exception:
|
||||
broken_page_indices.append(page_index)
|
||||
finally:
|
||||
close_pdfium_child(page)
|
||||
finally:
|
||||
close_pdfium_document(pdf_doc)
|
||||
|
||||
return loadable_page_indices, broken_page_indices
|
||||
|
||||
|
||||
def rewrite_pdf_bytes_with_pdfium(
|
||||
src_pdf_bytes: bytes,
|
||||
start_page_id: int = 0,
|
||||
@@ -79,7 +145,8 @@ def rewrite_pdf_bytes_with_pdfium(
|
||||
output_doc.save(output_buffer)
|
||||
return output_buffer.getvalue()
|
||||
finally:
|
||||
if output_doc is not None:
|
||||
close_pdfium_document(output_doc)
|
||||
if pdf_doc is not None:
|
||||
close_pdfium_document(pdf_doc)
|
||||
close_pdfium_objects_safely(
|
||||
output_doc,
|
||||
pdf_doc,
|
||||
owner="rewrite_pdf_bytes_with_pdfium",
|
||||
)
|
||||
|
||||
@@ -12,7 +12,7 @@ from mineru.utils.boxbase import calculate_overlap_area_in_bbox1_area_ratio
|
||||
from mineru.utils.enum_class import BlockType, ContentType
|
||||
from mineru.utils.pdf_image_tools import get_crop_img
|
||||
from mineru.utils.pdf_text_tool import get_lines_from_chars, get_page_chars
|
||||
from mineru.utils.pdfium_guard import pdfium_guard
|
||||
from mineru.utils.pdfium_guard import close_pdfium_child, pdfium_guard
|
||||
|
||||
MAX_NATIVE_TEXT_CHARS_PER_PAGE = 65535
|
||||
|
||||
@@ -35,91 +35,94 @@ def txt_spans_extract(pdf_page, spans, pil_img, scale, all_bboxes, all_discarded
|
||||
page_char_count = None
|
||||
textpage = None
|
||||
try:
|
||||
with pdfium_guard():
|
||||
textpage = pdf_page.get_textpage()
|
||||
page_char_count = textpage.count_chars()
|
||||
except Exception as exc:
|
||||
logger.debug(f"Failed to get page char count before txt extraction: {exc}")
|
||||
try:
|
||||
with pdfium_guard():
|
||||
textpage = pdf_page.get_textpage()
|
||||
page_char_count = textpage.count_chars()
|
||||
except Exception as exc:
|
||||
logger.debug(f"Failed to get page char count before txt extraction: {exc}")
|
||||
|
||||
if page_char_count is not None and page_char_count > MAX_NATIVE_TEXT_CHARS_PER_PAGE:
|
||||
logger.info(
|
||||
"Fallback to post-OCR in txt_spans_extract due to high char count: "
|
||||
f"count_chars={page_char_count}"
|
||||
if page_char_count is not None and page_char_count > MAX_NATIVE_TEXT_CHARS_PER_PAGE:
|
||||
logger.info(
|
||||
"Fallback to post-OCR in txt_spans_extract due to high char count: "
|
||||
f"count_chars={page_char_count}"
|
||||
)
|
||||
need_ocr_spans = [
|
||||
span for span in spans if span.get('type') == ContentType.TEXT
|
||||
]
|
||||
return _prepare_post_ocr_spans(need_ocr_spans, spans, pil_img, scale)
|
||||
|
||||
page_chars = get_page_chars(
|
||||
pdf_page,
|
||||
textpage=textpage,
|
||||
page_char_count=page_char_count,
|
||||
)
|
||||
need_ocr_spans = [
|
||||
span for span in spans if span.get('type') == ContentType.TEXT
|
||||
page_all_chars = [
|
||||
char for char in page_chars['chars']
|
||||
if _is_supported_rotation(char['rotation'])
|
||||
]
|
||||
return _prepare_post_ocr_spans(need_ocr_spans, spans, pil_img, scale)
|
||||
|
||||
page_chars = get_page_chars(
|
||||
pdf_page,
|
||||
textpage=textpage,
|
||||
page_char_count=page_char_count,
|
||||
)
|
||||
page_all_chars = [
|
||||
char for char in page_chars['chars']
|
||||
if _is_supported_rotation(char['rotation'])
|
||||
]
|
||||
# 计算所有sapn的高度的中位数
|
||||
span_height_list = []
|
||||
for span in spans:
|
||||
if span['type'] in [ContentType.TEXT]:
|
||||
span_height = span['bbox'][3] - span['bbox'][1]
|
||||
span['height'] = span_height
|
||||
span['width'] = span['bbox'][2] - span['bbox'][0]
|
||||
span_height_list.append(span_height)
|
||||
if len(span_height_list) == 0:
|
||||
return spans
|
||||
else:
|
||||
median_span_height = statistics.median(span_height_list)
|
||||
|
||||
# 计算所有sapn的高度的中位数
|
||||
span_height_list = []
|
||||
for span in spans:
|
||||
if span['type'] in [ContentType.TEXT]:
|
||||
span_height = span['bbox'][3] - span['bbox'][1]
|
||||
span['height'] = span_height
|
||||
span['width'] = span['bbox'][2] - span['bbox'][0]
|
||||
span_height_list.append(span_height)
|
||||
if len(span_height_list) == 0:
|
||||
return spans
|
||||
else:
|
||||
median_span_height = statistics.median(span_height_list)
|
||||
useful_spans = []
|
||||
unuseful_spans = []
|
||||
# 纵向span的两个特征:1. 高度超过多个line 2. 高宽比超过某个值
|
||||
vertical_spans = []
|
||||
for span in spans:
|
||||
if span['type'] in [ContentType.TEXT]:
|
||||
for block in all_bboxes + all_discarded_blocks:
|
||||
if block[7] in [BlockType.IMAGE_BODY, BlockType.TABLE_BODY, BlockType.INTERLINE_EQUATION]:
|
||||
continue
|
||||
if calculate_overlap_area_in_bbox1_area_ratio(span['bbox'], block[0:4]) > 0.5:
|
||||
if span['height'] > median_span_height * 2.3 and span['height'] > span['width'] * 2.3:
|
||||
vertical_spans.append(span)
|
||||
elif block in all_bboxes:
|
||||
useful_spans.append(span)
|
||||
else:
|
||||
unuseful_spans.append(span)
|
||||
break
|
||||
|
||||
useful_spans = []
|
||||
unuseful_spans = []
|
||||
# 纵向span的两个特征:1. 高度超过多个line 2. 高宽比超过某个值
|
||||
vertical_spans = []
|
||||
for span in spans:
|
||||
if span['type'] in [ContentType.TEXT]:
|
||||
for block in all_bboxes + all_discarded_blocks:
|
||||
if block[7] in [BlockType.IMAGE_BODY, BlockType.TABLE_BODY, BlockType.INTERLINE_EQUATION]:
|
||||
continue
|
||||
if calculate_overlap_area_in_bbox1_area_ratio(span['bbox'], block[0:4]) > 0.5:
|
||||
if span['height'] > median_span_height * 2.3 and span['height'] > span['width'] * 2.3:
|
||||
vertical_spans.append(span)
|
||||
elif block in all_bboxes:
|
||||
useful_spans.append(span)
|
||||
else:
|
||||
unuseful_spans.append(span)
|
||||
break
|
||||
"""垂直的span框直接用line进行填充"""
|
||||
if len(vertical_spans) > 0:
|
||||
page_all_lines = [
|
||||
line for line in get_lines_from_chars(page_chars['chars'])
|
||||
if _is_supported_rotation(line['rotation'])
|
||||
]
|
||||
for pdfium_line in page_all_lines:
|
||||
for span in vertical_spans:
|
||||
if calculate_overlap_area_in_bbox1_area_ratio(pdfium_line['bbox'].bbox, span['bbox']) > 0.5:
|
||||
for pdfium_span in pdfium_line['spans']:
|
||||
span['content'] += pdfium_span['text']
|
||||
break
|
||||
|
||||
"""垂直的span框直接用line进行填充"""
|
||||
if len(vertical_spans) > 0:
|
||||
page_all_lines = [
|
||||
line for line in get_lines_from_chars(page_chars['chars'])
|
||||
if _is_supported_rotation(line['rotation'])
|
||||
]
|
||||
for pdfium_line in page_all_lines:
|
||||
for span in vertical_spans:
|
||||
if calculate_overlap_area_in_bbox1_area_ratio(pdfium_line['bbox'].bbox, span['bbox']) > 0.5:
|
||||
for pdfium_span in pdfium_line['spans']:
|
||||
span['content'] += pdfium_span['text']
|
||||
break
|
||||
if len(span['content']) == 0:
|
||||
spans.remove(span)
|
||||
|
||||
for span in vertical_spans:
|
||||
if len(span['content']) == 0:
|
||||
spans.remove(span)
|
||||
"""水平的span框先用char填充,再用ocr填充空的span框"""
|
||||
new_spans = []
|
||||
|
||||
"""水平的span框先用char填充,再用ocr填充空的span框"""
|
||||
new_spans = []
|
||||
for span in useful_spans + unuseful_spans:
|
||||
if span['type'] in [ContentType.TEXT]:
|
||||
span['chars'] = []
|
||||
new_spans.append(span)
|
||||
|
||||
for span in useful_spans + unuseful_spans:
|
||||
if span['type'] in [ContentType.TEXT]:
|
||||
span['chars'] = []
|
||||
new_spans.append(span)
|
||||
need_ocr_spans = fill_char_in_spans(new_spans, page_all_chars, median_span_height)
|
||||
|
||||
need_ocr_spans = fill_char_in_spans(new_spans, page_all_chars, median_span_height)
|
||||
|
||||
return _prepare_post_ocr_spans(need_ocr_spans, spans, pil_img, scale)
|
||||
return _prepare_post_ocr_spans(need_ocr_spans, spans, pil_img, scale)
|
||||
finally:
|
||||
close_pdfium_child(textpage)
|
||||
|
||||
|
||||
def _is_supported_rotation(rotation) -> bool:
|
||||
|
||||
@@ -1,9 +1,12 @@
|
||||
# Copyright (c) Opendatalab. All rights reserved.
|
||||
from __future__ import annotations
|
||||
|
||||
import time
|
||||
from copy import deepcopy
|
||||
from typing import Any
|
||||
|
||||
from loguru import logger
|
||||
|
||||
from mineru.utils.config_reader import get_llm_aided_config
|
||||
from mineru.utils.llm_aided import llm_aided_title
|
||||
|
||||
@@ -29,7 +32,15 @@ def _resolve_title_aided_config() -> dict[str, Any] | None:
|
||||
def apply_title_leveling_to_pdf_info(pdf_info: list[dict[str, Any]]):
|
||||
title_aided_config = _resolve_title_aided_config()
|
||||
if title_aided_config:
|
||||
llm_aided_title(pdf_info, title_aided_config)
|
||||
start_time = time.perf_counter()
|
||||
success = False
|
||||
try:
|
||||
llm_aided_title(pdf_info, title_aided_config)
|
||||
success = True
|
||||
finally:
|
||||
elapsed = time.perf_counter() - start_time
|
||||
status = "finished" if success else "failed"
|
||||
logger.info(f"title leveling {status}, cost: {elapsed:.2f}s")
|
||||
|
||||
|
||||
def finalize_client_side_middle_json(middle_json: dict[str, Any]) -> dict[str, Any]:
|
||||
|
||||
Reference in New Issue
Block a user