From 41931b861c17d3f02fe51cca246db92700a363ec Mon Sep 17 00:00:00 2001 From: myhloli Date: Tue, 9 Jun 2026 16:29:51 +0800 Subject: [PATCH] feat: enhance PDF character processing with deduplication and control character filtering --- mineru/utils/pdf_text_tool.py | 64 +++++++++++++++++++++++++++++- mineru/utils/span_pre_proc.py | 74 ++++++++++++++++++++++------------- 2 files changed, 109 insertions(+), 29 deletions(-) diff --git a/mineru/utils/pdf_text_tool.py b/mineru/utils/pdf_text_tool.py index 7cb495fd..c24ca081 100644 --- a/mineru/utils/pdf_text_tool.py +++ b/mineru/utils/pdf_text_tool.py @@ -1,6 +1,6 @@ # Copyright (c) Opendatalab. All rights reserved. import math -from typing import List +from typing import Any, List import pypdfium2 as pdfium from pdftext.pdf.chars import deduplicate_chars, get_chars @@ -8,6 +8,8 @@ from pdftext.pdf.pages import assign_scripts, get_blocks, get_lines, get_spans from mineru.utils.pdfium_guard import close_pdfium_child, pdfium_guard +NEAR_IDENTICAL_CHAR_BBOX_TOLERANCE = 1.0 + def get_page( page: pdfium.PdfPage, @@ -32,6 +34,65 @@ def get_page( } +def _get_char_bbox_coords(char: dict[str, Any]) -> tuple[float, ...]: + """统一提取字符 bbox 坐标,兼容 pdftext Bbox 对象和普通 list。""" + bbox = char.get("bbox") + bbox_coords = getattr(bbox, "bbox", bbox) + return tuple(float(coord) for coord in bbox_coords) + + +def _get_visible_char_signature( + char: dict[str, Any], +) -> tuple[str, tuple[Any, Any, Any, Any], float]: + """生成可见字符去重签名,不把 bbox 放入签名以便单独做近重合判断。""" + font = char.get("font") or {} + font_key = ( + font.get("name"), + font.get("flags"), + font.get("size"), + font.get("weight"), + ) + rotation_key = round(float(char.get("rotation") or 0.0), 3) + return char.get("char", ""), font_key, rotation_key + + +def _is_near_identical_bbox( + bbox_a: tuple[float, ...], + bbox_b: tuple[float, ...], +) -> bool: + """判断两个字符 bbox 是否属于同一视觉位置的一点内抖动。""" + return all( + abs(coord_a - coord_b) <= NEAR_IDENTICAL_CHAR_BBOX_TOLERANCE + for coord_a, coord_b in zip(bbox_a, bbox_b) + ) + + +def _deduplicate_near_identical_chars( + chars: list[dict[str, Any]], +) -> list[dict[str, Any]]: + """清理 PDFium 文本层边界处同字符、同位置的重复可见字符。""" + seen_visible_char_bboxes = {} + deduplicated_chars = [] + + for char in chars: + text = char.get("char", "") + if not text or text.isspace(): + deduplicated_chars.append(char) + continue + + visible_char_key = _get_visible_char_signature(char) + bbox_coords = _get_char_bbox_coords(char) + if any( + _is_near_identical_bbox(bbox_coords, seen_bbox) + for seen_bbox in seen_visible_char_bboxes.get(visible_char_key, []) + ): + continue + seen_visible_char_bboxes.setdefault(visible_char_key, []).append(bbox_coords) + deduplicated_chars.append(char) + + return deduplicated_chars + + def get_page_chars( page: pdfium.PdfPage, textpage=None, @@ -60,6 +121,7 @@ def get_page_chars( chars = deduplicate_chars( get_chars(textpage, page_bbox, page_rotation, quote_loosebox) ) + chars = _deduplicate_near_identical_chars(chars) finally: if owns_textpage: close_pdfium_child(textpage) diff --git a/mineru/utils/span_pre_proc.py b/mineru/utils/span_pre_proc.py index cdb77694..b864a571 100644 --- a/mineru/utils/span_pre_proc.py +++ b/mineru/utils/span_pre_proc.py @@ -296,8 +296,13 @@ def fill_char_in_spans(spans, all_chars, median_span_height): return need_ocr_spans -LINE_STOP_FLAG = ('.', '!', '?', '。', '!', '?', ')', ')', '"', '”', ':', ':', ';', ';', ']', '】', '}', '}', '>', '》', '、', ',', ',', '-', '—', '–',) -LINE_START_FLAG = ('(', '(', '"', '“', '【', '{', '《', '<', '「', '『', '【', '[',) +LINE_STOP_FLAG = ( + '.', '!', '?', '。', '!', '?', ')', ')', '"', '”', ':', ':', ';', + ';', ']', '】', '}', '}', '>', '》', '、', ',', ',', '-', '—', '–', +) +LINE_START_FLAG = ( + '(', '(', '"', '“', '【', '{', '《', '<', '「', '『', '【', '[', +) Span_Height_Ratio = 0.33 # 字符的中轴和span的中轴高度差不能超过1/3span高度 SCRIPT_BODY_HEIGHT_RATIO = 0.9 @@ -386,7 +391,8 @@ def calculate_char_in_span(char_bbox, span_bbox, char, span_height_ratio=Span_He if ( span_bbox[0] < char_center_x < span_bbox[2] and span_bbox[1] < char_center_y < span_bbox[3] - and abs(char_center_y - span_center_y) < span_height * span_height_ratio # 字符的中轴和span的中轴高度差不能超过Span_Height_Ratio + # 字符的中轴和span的中轴高度差不能超过Span_Height_Ratio + and abs(char_center_y - span_center_y) < span_height * span_height_ratio ): return True else: @@ -520,6 +526,14 @@ def _wrap_script_runs(role_text_parts): return ''.join(wrapped_parts) +def _remove_control_line_break_chars(chars): + """过滤 PDFium 文本片段边界控制换行,避免其参与字符间距补空格。""" + return [ + char for char in chars + if char.get('char') not in {'\r', '\n'} + ] + + def chars_to_content(span): # 检查span中的char是否为空 if len(span['chars']) != 0: @@ -531,34 +545,38 @@ def chars_to_content(span): ): chars = sorted(chars, key=lambda x: x['char_idx']) - char_metrics = _get_char_bbox_metrics_list(chars) - # Calculate the width of each character - char_widths = [metrics['width'] for metrics in char_metrics] - # Calculate the median width - median_width = statistics.median(char_widths) - script_roles = _classify_char_script_roles(chars, char_metrics) + chars = _remove_control_line_break_chars(chars) + if len(chars) == 0: + span['content'] = '' + else: + char_metrics = _get_char_bbox_metrics_list(chars) + # Calculate the width of each character + char_widths = [metrics['width'] for metrics in char_metrics] + # Calculate the median width + median_width = statistics.median(char_widths) + script_roles = _classify_char_script_roles(chars, char_metrics) - role_text_parts = [] - for idx, char1 in enumerate(chars): - char2 = chars[idx + 1] if idx + 1 < len(chars) else None - role1 = script_roles[idx] - role2 = script_roles[idx + 1] if char2 else None + role_text_parts = [] + for idx, char1 in enumerate(chars): + char2 = chars[idx + 1] if idx + 1 < len(chars) else None + role1 = script_roles[idx] + role2 = script_roles[idx + 1] if char2 else None - # 如果下一个char的x0和上一个char的x1距离超过0.25个字符宽度,则需要在中间插入一个空格 - role_text_parts.append((role1, char1['char'])) - if ( - char2 - and char2['bbox'][0] - char1['bbox'][2] > median_width * 0.25 - and char1['char'] != ' ' - and char2['char'] != ' ' - ): - space_role = role1 if role1 == role2 else 'body' - role_text_parts.append((space_role, ' ')) + # 如果下一个char的x0和上一个char的x1距离超过0.25个字符宽度,则需要在中间插入一个空格 + role_text_parts.append((role1, char1['char'])) + if ( + char2 + and char2['bbox'][0] - char1['bbox'][2] > median_width * 0.25 + and char1['char'] != ' ' + and char2['char'] != ' ' + ): + space_role = role1 if role1 == role2 else 'body' + role_text_parts.append((space_role, ' ')) - content = _wrap_script_runs(role_text_parts) - content = __replace_unicode(content) - content = __replace_ligatures(content) - span['content'] = content.strip() + content = _wrap_script_runs(role_text_parts) + content = __replace_unicode(content) + content = __replace_ligatures(content) + span['content'] = content.strip() del span['chars']