feat: enhance PDF character processing with deduplication and control character filtering

This commit is contained in:
myhloli
2026-06-09 16:29:51 +08:00
parent 1725e25707
commit 41931b861c
2 changed files with 109 additions and 29 deletions
+63 -1
View File
@@ -1,6 +1,6 @@
# Copyright (c) Opendatalab. All rights reserved.
import math
from typing import List
from typing import Any, List
import pypdfium2 as pdfium
from pdftext.pdf.chars import deduplicate_chars, get_chars
@@ -8,6 +8,8 @@ from pdftext.pdf.pages import assign_scripts, get_blocks, get_lines, get_spans
from mineru.utils.pdfium_guard import close_pdfium_child, pdfium_guard
NEAR_IDENTICAL_CHAR_BBOX_TOLERANCE = 1.0
def get_page(
page: pdfium.PdfPage,
@@ -32,6 +34,65 @@ def get_page(
}
def _get_char_bbox_coords(char: dict[str, Any]) -> tuple[float, ...]:
"""统一提取字符 bbox 坐标,兼容 pdftext Bbox 对象和普通 list。"""
bbox = char.get("bbox")
bbox_coords = getattr(bbox, "bbox", bbox)
return tuple(float(coord) for coord in bbox_coords)
def _get_visible_char_signature(
char: dict[str, Any],
) -> tuple[str, tuple[Any, Any, Any, Any], float]:
"""生成可见字符去重签名,不把 bbox 放入签名以便单独做近重合判断。"""
font = char.get("font") or {}
font_key = (
font.get("name"),
font.get("flags"),
font.get("size"),
font.get("weight"),
)
rotation_key = round(float(char.get("rotation") or 0.0), 3)
return char.get("char", ""), font_key, rotation_key
def _is_near_identical_bbox(
bbox_a: tuple[float, ...],
bbox_b: tuple[float, ...],
) -> bool:
"""判断两个字符 bbox 是否属于同一视觉位置的一点内抖动。"""
return all(
abs(coord_a - coord_b) <= NEAR_IDENTICAL_CHAR_BBOX_TOLERANCE
for coord_a, coord_b in zip(bbox_a, bbox_b)
)
def _deduplicate_near_identical_chars(
chars: list[dict[str, Any]],
) -> list[dict[str, Any]]:
"""清理 PDFium 文本层边界处同字符、同位置的重复可见字符。"""
seen_visible_char_bboxes = {}
deduplicated_chars = []
for char in chars:
text = char.get("char", "")
if not text or text.isspace():
deduplicated_chars.append(char)
continue
visible_char_key = _get_visible_char_signature(char)
bbox_coords = _get_char_bbox_coords(char)
if any(
_is_near_identical_bbox(bbox_coords, seen_bbox)
for seen_bbox in seen_visible_char_bboxes.get(visible_char_key, [])
):
continue
seen_visible_char_bboxes.setdefault(visible_char_key, []).append(bbox_coords)
deduplicated_chars.append(char)
return deduplicated_chars
def get_page_chars(
page: pdfium.PdfPage,
textpage=None,
@@ -60,6 +121,7 @@ def get_page_chars(
chars = deduplicate_chars(
get_chars(textpage, page_bbox, page_rotation, quote_loosebox)
)
chars = _deduplicate_near_identical_chars(chars)
finally:
if owns_textpage:
close_pdfium_child(textpage)
+46 -28
View File
@@ -296,8 +296,13 @@ def fill_char_in_spans(spans, all_chars, median_span_height):
return need_ocr_spans
LINE_STOP_FLAG = ('.', '!', '?', '。', '!', '?', ')', ')', '"', '”', ':', ':', ';', ';', ']', '】', '}', '}', '>', '》', '、', ',', ',', '-', '—', '–',)
LINE_START_FLAG = ('(', '(', '"', '“', '【', '{', '《', '<', '「', '『', '【', '[',)
LINE_STOP_FLAG = (
'.', '!', '?', '。', '!', '?', ')', ')', '"', '”', ':', ':', ';',
';', ']', '】', '}', '}', '>', '》', '、', ',', ',', '-', '—', '–',
)
LINE_START_FLAG = (
'(', '(', '"', '“', '【', '{', '《', '<', '「', '『', '【', '[',
)
Span_Height_Ratio = 0.33 # 字符的中轴和span的中轴高度差不能超过1/3span高度
SCRIPT_BODY_HEIGHT_RATIO = 0.9
@@ -386,7 +391,8 @@ def calculate_char_in_span(char_bbox, span_bbox, char, span_height_ratio=Span_He
if (
span_bbox[0] < char_center_x < span_bbox[2]
and span_bbox[1] < char_center_y < span_bbox[3]
and abs(char_center_y - span_center_y) < span_height * span_height_ratio # 字符的中轴和span的中轴高度差不能超过Span_Height_Ratio
# 字符的中轴和span的中轴高度差不能超过Span_Height_Ratio
and abs(char_center_y - span_center_y) < span_height * span_height_ratio
):
return True
else:
@@ -520,6 +526,14 @@ def _wrap_script_runs(role_text_parts):
return ''.join(wrapped_parts)
def _remove_control_line_break_chars(chars):
"""过滤 PDFium 文本片段边界控制换行,避免其参与字符间距补空格。"""
return [
char for char in chars
if char.get('char') not in {'\r', '\n'}
]
def chars_to_content(span):
# 检查span中的char是否为空
if len(span['chars']) != 0:
@@ -531,34 +545,38 @@ def chars_to_content(span):
):
chars = sorted(chars, key=lambda x: x['char_idx'])
char_metrics = _get_char_bbox_metrics_list(chars)
# Calculate the width of each character
char_widths = [metrics['width'] for metrics in char_metrics]
# Calculate the median width
median_width = statistics.median(char_widths)
script_roles = _classify_char_script_roles(chars, char_metrics)
chars = _remove_control_line_break_chars(chars)
if len(chars) == 0:
span['content'] = ''
else:
char_metrics = _get_char_bbox_metrics_list(chars)
# Calculate the width of each character
char_widths = [metrics['width'] for metrics in char_metrics]
# Calculate the median width
median_width = statistics.median(char_widths)
script_roles = _classify_char_script_roles(chars, char_metrics)
role_text_parts = []
for idx, char1 in enumerate(chars):
char2 = chars[idx + 1] if idx + 1 < len(chars) else None
role1 = script_roles[idx]
role2 = script_roles[idx + 1] if char2 else None
role_text_parts = []
for idx, char1 in enumerate(chars):
char2 = chars[idx + 1] if idx + 1 < len(chars) else None
role1 = script_roles[idx]
role2 = script_roles[idx + 1] if char2 else None
# 如果下一个char的x0和上一个char的x1距离超过0.25个字符宽度,则需要在中间插入一个空格
role_text_parts.append((role1, char1['char']))
if (
char2
and char2['bbox'][0] - char1['bbox'][2] > median_width * 0.25
and char1['char'] != ' '
and char2['char'] != ' '
):
space_role = role1 if role1 == role2 else 'body'
role_text_parts.append((space_role, ' '))
# 如果下一个char的x0和上一个char的x1距离超过0.25个字符宽度,则需要在中间插入一个空格
role_text_parts.append((role1, char1['char']))
if (
char2
and char2['bbox'][0] - char1['bbox'][2] > median_width * 0.25
and char1['char'] != ' '
and char2['char'] != ' '
):
space_role = role1 if role1 == role2 else 'body'
role_text_parts.append((space_role, ' '))
content = _wrap_script_runs(role_text_parts)
content = __replace_unicode(content)
content = __replace_ligatures(content)
span['content'] = content.strip()
content = _wrap_script_runs(role_text_parts)
content = __replace_unicode(content)
content = __replace_ligatures(content)
span['content'] = content.strip()
del span['chars']