mirror of
https://github.com/opendatalab/MinerU.git
synced 2026-09-24 23:10:23 +08:00
feat: enhance PDF character processing with deduplication and control character filtering
This commit is contained in:
@@ -1,6 +1,6 @@
|
||||
# Copyright (c) Opendatalab. All rights reserved.
|
||||
import math
|
||||
from typing import List
|
||||
from typing import Any, List
|
||||
|
||||
import pypdfium2 as pdfium
|
||||
from pdftext.pdf.chars import deduplicate_chars, get_chars
|
||||
@@ -8,6 +8,8 @@ from pdftext.pdf.pages import assign_scripts, get_blocks, get_lines, get_spans
|
||||
|
||||
from mineru.utils.pdfium_guard import close_pdfium_child, pdfium_guard
|
||||
|
||||
NEAR_IDENTICAL_CHAR_BBOX_TOLERANCE = 1.0
|
||||
|
||||
|
||||
def get_page(
|
||||
page: pdfium.PdfPage,
|
||||
@@ -32,6 +34,65 @@ def get_page(
|
||||
}
|
||||
|
||||
|
||||
def _get_char_bbox_coords(char: dict[str, Any]) -> tuple[float, ...]:
|
||||
"""统一提取字符 bbox 坐标,兼容 pdftext Bbox 对象和普通 list。"""
|
||||
bbox = char.get("bbox")
|
||||
bbox_coords = getattr(bbox, "bbox", bbox)
|
||||
return tuple(float(coord) for coord in bbox_coords)
|
||||
|
||||
|
||||
def _get_visible_char_signature(
|
||||
char: dict[str, Any],
|
||||
) -> tuple[str, tuple[Any, Any, Any, Any], float]:
|
||||
"""生成可见字符去重签名,不把 bbox 放入签名以便单独做近重合判断。"""
|
||||
font = char.get("font") or {}
|
||||
font_key = (
|
||||
font.get("name"),
|
||||
font.get("flags"),
|
||||
font.get("size"),
|
||||
font.get("weight"),
|
||||
)
|
||||
rotation_key = round(float(char.get("rotation") or 0.0), 3)
|
||||
return char.get("char", ""), font_key, rotation_key
|
||||
|
||||
|
||||
def _is_near_identical_bbox(
|
||||
bbox_a: tuple[float, ...],
|
||||
bbox_b: tuple[float, ...],
|
||||
) -> bool:
|
||||
"""判断两个字符 bbox 是否属于同一视觉位置的一点内抖动。"""
|
||||
return all(
|
||||
abs(coord_a - coord_b) <= NEAR_IDENTICAL_CHAR_BBOX_TOLERANCE
|
||||
for coord_a, coord_b in zip(bbox_a, bbox_b)
|
||||
)
|
||||
|
||||
|
||||
def _deduplicate_near_identical_chars(
|
||||
chars: list[dict[str, Any]],
|
||||
) -> list[dict[str, Any]]:
|
||||
"""清理 PDFium 文本层边界处同字符、同位置的重复可见字符。"""
|
||||
seen_visible_char_bboxes = {}
|
||||
deduplicated_chars = []
|
||||
|
||||
for char in chars:
|
||||
text = char.get("char", "")
|
||||
if not text or text.isspace():
|
||||
deduplicated_chars.append(char)
|
||||
continue
|
||||
|
||||
visible_char_key = _get_visible_char_signature(char)
|
||||
bbox_coords = _get_char_bbox_coords(char)
|
||||
if any(
|
||||
_is_near_identical_bbox(bbox_coords, seen_bbox)
|
||||
for seen_bbox in seen_visible_char_bboxes.get(visible_char_key, [])
|
||||
):
|
||||
continue
|
||||
seen_visible_char_bboxes.setdefault(visible_char_key, []).append(bbox_coords)
|
||||
deduplicated_chars.append(char)
|
||||
|
||||
return deduplicated_chars
|
||||
|
||||
|
||||
def get_page_chars(
|
||||
page: pdfium.PdfPage,
|
||||
textpage=None,
|
||||
@@ -60,6 +121,7 @@ def get_page_chars(
|
||||
chars = deduplicate_chars(
|
||||
get_chars(textpage, page_bbox, page_rotation, quote_loosebox)
|
||||
)
|
||||
chars = _deduplicate_near_identical_chars(chars)
|
||||
finally:
|
||||
if owns_textpage:
|
||||
close_pdfium_child(textpage)
|
||||
|
||||
@@ -296,8 +296,13 @@ def fill_char_in_spans(spans, all_chars, median_span_height):
|
||||
return need_ocr_spans
|
||||
|
||||
|
||||
LINE_STOP_FLAG = ('.', '!', '?', '。', '!', '?', ')', ')', '"', '”', ':', ':', ';', ';', ']', '】', '}', '}', '>', '》', '、', ',', ',', '-', '—', '–',)
|
||||
LINE_START_FLAG = ('(', '(', '"', '“', '【', '{', '《', '<', '「', '『', '【', '[',)
|
||||
LINE_STOP_FLAG = (
|
||||
'.', '!', '?', '。', '!', '?', ')', ')', '"', '”', ':', ':', ';',
|
||||
';', ']', '】', '}', '}', '>', '》', '、', ',', ',', '-', '—', '–',
|
||||
)
|
||||
LINE_START_FLAG = (
|
||||
'(', '(', '"', '“', '【', '{', '《', '<', '「', '『', '【', '[',
|
||||
)
|
||||
|
||||
Span_Height_Ratio = 0.33 # 字符的中轴和span的中轴高度差不能超过1/3span高度
|
||||
SCRIPT_BODY_HEIGHT_RATIO = 0.9
|
||||
@@ -386,7 +391,8 @@ def calculate_char_in_span(char_bbox, span_bbox, char, span_height_ratio=Span_He
|
||||
if (
|
||||
span_bbox[0] < char_center_x < span_bbox[2]
|
||||
and span_bbox[1] < char_center_y < span_bbox[3]
|
||||
and abs(char_center_y - span_center_y) < span_height * span_height_ratio # 字符的中轴和span的中轴高度差不能超过Span_Height_Ratio
|
||||
# 字符的中轴和span的中轴高度差不能超过Span_Height_Ratio
|
||||
and abs(char_center_y - span_center_y) < span_height * span_height_ratio
|
||||
):
|
||||
return True
|
||||
else:
|
||||
@@ -520,6 +526,14 @@ def _wrap_script_runs(role_text_parts):
|
||||
return ''.join(wrapped_parts)
|
||||
|
||||
|
||||
def _remove_control_line_break_chars(chars):
|
||||
"""过滤 PDFium 文本片段边界控制换行,避免其参与字符间距补空格。"""
|
||||
return [
|
||||
char for char in chars
|
||||
if char.get('char') not in {'\r', '\n'}
|
||||
]
|
||||
|
||||
|
||||
def chars_to_content(span):
|
||||
# 检查span中的char是否为空
|
||||
if len(span['chars']) != 0:
|
||||
@@ -531,34 +545,38 @@ def chars_to_content(span):
|
||||
):
|
||||
chars = sorted(chars, key=lambda x: x['char_idx'])
|
||||
|
||||
char_metrics = _get_char_bbox_metrics_list(chars)
|
||||
# Calculate the width of each character
|
||||
char_widths = [metrics['width'] for metrics in char_metrics]
|
||||
# Calculate the median width
|
||||
median_width = statistics.median(char_widths)
|
||||
script_roles = _classify_char_script_roles(chars, char_metrics)
|
||||
chars = _remove_control_line_break_chars(chars)
|
||||
if len(chars) == 0:
|
||||
span['content'] = ''
|
||||
else:
|
||||
char_metrics = _get_char_bbox_metrics_list(chars)
|
||||
# Calculate the width of each character
|
||||
char_widths = [metrics['width'] for metrics in char_metrics]
|
||||
# Calculate the median width
|
||||
median_width = statistics.median(char_widths)
|
||||
script_roles = _classify_char_script_roles(chars, char_metrics)
|
||||
|
||||
role_text_parts = []
|
||||
for idx, char1 in enumerate(chars):
|
||||
char2 = chars[idx + 1] if idx + 1 < len(chars) else None
|
||||
role1 = script_roles[idx]
|
||||
role2 = script_roles[idx + 1] if char2 else None
|
||||
role_text_parts = []
|
||||
for idx, char1 in enumerate(chars):
|
||||
char2 = chars[idx + 1] if idx + 1 < len(chars) else None
|
||||
role1 = script_roles[idx]
|
||||
role2 = script_roles[idx + 1] if char2 else None
|
||||
|
||||
# 如果下一个char的x0和上一个char的x1距离超过0.25个字符宽度,则需要在中间插入一个空格
|
||||
role_text_parts.append((role1, char1['char']))
|
||||
if (
|
||||
char2
|
||||
and char2['bbox'][0] - char1['bbox'][2] > median_width * 0.25
|
||||
and char1['char'] != ' '
|
||||
and char2['char'] != ' '
|
||||
):
|
||||
space_role = role1 if role1 == role2 else 'body'
|
||||
role_text_parts.append((space_role, ' '))
|
||||
# 如果下一个char的x0和上一个char的x1距离超过0.25个字符宽度,则需要在中间插入一个空格
|
||||
role_text_parts.append((role1, char1['char']))
|
||||
if (
|
||||
char2
|
||||
and char2['bbox'][0] - char1['bbox'][2] > median_width * 0.25
|
||||
and char1['char'] != ' '
|
||||
and char2['char'] != ' '
|
||||
):
|
||||
space_role = role1 if role1 == role2 else 'body'
|
||||
role_text_parts.append((space_role, ' '))
|
||||
|
||||
content = _wrap_script_runs(role_text_parts)
|
||||
content = __replace_unicode(content)
|
||||
content = __replace_ligatures(content)
|
||||
span['content'] = content.strip()
|
||||
content = _wrap_script_runs(role_text_parts)
|
||||
content = __replace_unicode(content)
|
||||
content = __replace_ligatures(content)
|
||||
span['content'] = content.strip()
|
||||
|
||||
del span['chars']
|
||||
|
||||
|
||||
Reference in New Issue
Block a user