mirror of
https://github.com/opendatalab/MinerU.git
synced 2026-09-24 23:10:23 +08:00
Initial commit
This commit is contained in:
@@ -0,0 +1,19 @@
|
||||
# pdf_toolbox
|
||||
pdf 解析基础函数
|
||||
|
||||
|
||||
## pdf是否是文字类型/扫描类型的区分
|
||||
|
||||
```shell
|
||||
cat s3_pdf_path.example.pdf | parallel --colsep ' ' -j 10 "python pdf_meta_scan.py --s3-pdf-path {2} --s3-profile {1} >> {/}.jsonl"
|
||||
|
||||
find dir/to/jsonl/ -type f -name "*.jsonl" | parallel -j 10 "python pdf_classfy_by_type.py --json_file {} >> {/}.jsonl"
|
||||
|
||||
```
|
||||
|
||||
```shell
|
||||
# 如果单独运行脚本,合并到code-clean之后需要运行,参考如下:
|
||||
python -m pdf_meta_scan --s3-pdf-path "D:\pdf_files\内容排序测试_pdf\p3_图文混排 5.pdf" --s3-profile s2
|
||||
```
|
||||
|
||||
## pdf
|
||||
@@ -0,0 +1,73 @@
|
||||
# 最终版:把那种text_block有重叠,且inline_formula位置在重叠部分的,认定整个页面都有问题,所有的inline_formula都改成no_check
|
||||
from libs.commons import fitz
|
||||
|
||||
|
||||
def check_inline_formula(page, inline_formula_boxes):
|
||||
"""
|
||||
:param page :fitz读取的当前页的内容
|
||||
:param inline_formula_boxes: list类型,每一个元素是一个元祖 (L, U, R, D)
|
||||
|
||||
:return: inline_formula_check: list类型,每一个元素是一个类别,其顺序对应输入的inline_formula_boxes,给每个行内公式打一个标签,包括:
|
||||
- nocheck_inline_formula:这个公式框没有与任何span相交,有可能存在问题
|
||||
- wrong_text_block:这个公式框同时存在多个block里,可能页面的text block存在问题
|
||||
- false_inline_formula:只涉及一个span并且只占据这个span的小部分面积,判断可能不是公式
|
||||
- true_inline_formula:两种情况判断为公式,一是横跨多个span,二是只涉及一个span但是几乎占据了这个span大部分的面积
|
||||
"""
|
||||
|
||||
# count = defaultdict(int)
|
||||
## ------------------------ Text --------------------------------------------
|
||||
blocks = page.get_text(
|
||||
"dict",
|
||||
flags=fitz.TEXTFLAGS_TEXT,
|
||||
#clip=clip,
|
||||
)["blocks"]
|
||||
|
||||
# iterate over the bboxes
|
||||
inline_formula_check = []
|
||||
for result in inline_formula_boxes:
|
||||
(x1, y1, x2, y2) = (result[0], result[1], result[2], result[3])
|
||||
## 逐个block##
|
||||
in_block = 0
|
||||
for bbox in blocks:
|
||||
# image = cv2.rectangle(image, (int(bbox['bbox'][0]), int(bbox['bbox'][1])), (int(bbox['bbox'][2]), int(bbox['bbox'][3])), (0, 255, 0), 1)
|
||||
if (y1 >= bbox['bbox'][1] and y2 <= bbox['bbox'][3]) and (x1 >= bbox['bbox'][0] and x2 <= bbox['bbox'][2]): # 判定公式在哪一个block
|
||||
in_block += 1
|
||||
intersect = []
|
||||
# ## 逐个span###
|
||||
for line in bbox['lines']:
|
||||
if line['bbox'][1] <= ((y2 - y1) / 2) + y1 <= line['bbox'][3]: # 判断公式在哪一行
|
||||
for item in line['spans']:
|
||||
(t_x1, t_y1, t_x2, t_y2) = item['bbox']
|
||||
if not ((t_x1 < x1 and t_x2 < x1) or (t_x1 > x2 and t_x2 > x2) or (t_y1 < y1 and t_y2 < y1) or (t_y1 > y2 and t_y2 > y2)): # 判断是否相交
|
||||
intersect.append(item['bbox'])
|
||||
# image = cv2.rectangle(image, (int(t_x1), int(t_y1)), (int(t_x2), int(t_y2)), (0, 255, 0), 1) # 可视化涉及到的span
|
||||
|
||||
# 可视化公式的分类
|
||||
if len(intersect) == 0: # 没有与任何一个span有相交,这个span或者这个inline_formula_box可能有问题
|
||||
# print(f'Wrong location, check {img_path}')
|
||||
inline_formula_check_result = "nocheck_inline_formula"
|
||||
# count['not_in_line'] += 1
|
||||
elif len(intersect) == 1:
|
||||
if abs((intersect[0][2] - intersect[0][0]) - (x2 - x1)) < (x2 - x1)*0.5: # 只涉及一个span但是几乎占据了这个span大部分的面积,判定为公式
|
||||
# image = cv2.rectangle(image, (int(x1), int(y1)), (int(x2), int(y2)), (0, 255, 0), 1)
|
||||
inline_formula_check_result = "true_inline_formula"
|
||||
# count['one_span_large'] += 1
|
||||
else: # 只涉及一个span并且只占据这个span的小部分面积,判断可能不是公式
|
||||
# image = cv2.rectangle(image, (int(x1), int(y1)), (int(x2), int(y2)), (0, 0, 255), 1)
|
||||
inline_formula_check_result = "false_inline_formula"
|
||||
# count['fail'] += 1
|
||||
else: # 横跨多个span,判定为公式
|
||||
# image = cv2.rectangle(image, (int(x1), int(y1)), (int(x2), int(y2)), (255, 0, 0), 1)
|
||||
inline_formula_check_result = "true_inline_formula"
|
||||
# count['multi_span'] += 1
|
||||
|
||||
if in_block == 0: # 这个公式没有在任何的block里,这个公式可能有问题
|
||||
# image = cv2.rectangle(image, (int(x1), int(y1)), (int(x2), int(y2)), (255, 255, 0), 1)
|
||||
inline_formula_check_result = "nocheck_inline_formula"
|
||||
# count['not_in_block'] += 1
|
||||
elif in_block > 1: # 这个公式存在于多个block里,这个页面可能有问题
|
||||
inline_formula_check_result = "wrong_text_block"
|
||||
|
||||
inline_formula_check.append(inline_formula_check_result)
|
||||
|
||||
return inline_formula_check
|
||||
+24
@@ -0,0 +1,24 @@
|
||||
import json
|
||||
import os
|
||||
from tqdm import tqdm
|
||||
|
||||
from libs.commons import join_path
|
||||
|
||||
with open('/mnt/petrelfs/share_data/ouyanglinke/OCR/OCR_validation_dataset.json', 'r') as f:
|
||||
samples = json.load(f)
|
||||
|
||||
pdf_model_dir = 's3://llm-pdf-text/eval_1k/layout_res/'
|
||||
|
||||
labels = []
|
||||
det_res = []
|
||||
edit_distance_list = []
|
||||
for sample in tqdm(samples):
|
||||
pdf_name = sample['pdf_name']
|
||||
page_num = sample['page']
|
||||
pdf_model_path = join_path(pdf_model_dir, pdf_name)
|
||||
model_output_json = join_path(pdf_model_path, f"page_{page_num}.json") # 模型输出的页面编号从1开始的
|
||||
save_root_path = '/mnt/petrelfs/share_data/ouyanglinke/OCR/OCR_val_docxchain/'
|
||||
save_path = join_path(save_root_path, pdf_name)
|
||||
os.makedirs(save_path, exist_ok=True)
|
||||
# print("s3c cp {} {}".format(model_output_json, save_path))
|
||||
os.system("aws --profile langchao --endpoint-url=http://10.140.85.161:80 s3 cp {} {}".format(model_output_json, save_path))
|
||||
@@ -0,0 +1,20 @@
|
||||
from libs.commons import fitz # PyMuPDF
|
||||
|
||||
# PDF文件路径
|
||||
pdf_path = "D:\\project\\20231108code-clean\\code-clean\\tmp\\unittest\\download-pdfs\\scihub\\scihub_53700000\\libgen.scimag53724000-53724999.zip_10.1097\\00129191-200509000-00018.pdf"
|
||||
|
||||
doc = fitz.open(pdf_path) # Open the PDF
|
||||
# 你的数据
|
||||
data = [[[-2, 0, 603, 80, 24]], [[-3, 0, 602, 80, 24]]]
|
||||
|
||||
# 对每个页面进行处理
|
||||
for i, page in enumerate(doc):
|
||||
# 获取当前页面的数据
|
||||
page_data = data[i]
|
||||
for img in page_data:
|
||||
x0, y0, x1, y1, _ = img
|
||||
rect_coords = fitz.Rect(x0, y0, x1, y1) # Define the rectangle
|
||||
page.draw_rect(rect_coords, color=(1, 0, 0), fill=None, width=1.5, overlay=True) # Draw the rectangle
|
||||
|
||||
# Save the PDF
|
||||
doc.save("D:\\project\\20231108code-clean\\code-clean\\tmp\\unittest\\download-pdfs\\scihub\\scihub_53700000\\libgen.scimag53724000-53724999.zip_10.1097\\00129191-200509000-00018_new.pdf")
|
||||
File diff suppressed because one or more lines are too long
@@ -0,0 +1,371 @@
|
||||
"""
|
||||
输入: s3路径,每行一个
|
||||
输出: pdf文件元信息,包括每一页上的所有图片的长宽高,bbox位置
|
||||
"""
|
||||
import math
|
||||
import sys
|
||||
import click
|
||||
|
||||
from libs.commons import read_file, mymax, get_top_percent_list
|
||||
import json
|
||||
from libs.commons import fitz
|
||||
from loguru import logger
|
||||
from collections import Counter
|
||||
|
||||
from libs.drop_reason import DropReason
|
||||
from libs.language import detect_lang
|
||||
|
||||
scan_max_page = 50
|
||||
junk_limit_min = 10
|
||||
|
||||
|
||||
def calculate_max_image_area_per_page(result:list, page_width_pts, page_height_pts):
|
||||
max_image_area_per_page = [mymax([(x1 - x0) * (y1 - y0) for x0, y0, x1, y1, _ in page_img_sz]) for page_img_sz in
|
||||
result]
|
||||
page_area = int(page_width_pts) * int(page_height_pts)
|
||||
max_image_area_per_page = [area / page_area for area in max_image_area_per_page]
|
||||
max_image_area_per_page = [area for area in max_image_area_per_page if area > 0.6]
|
||||
return max_image_area_per_page
|
||||
|
||||
def process_image(page, junk_img_bojids=[]):
|
||||
page_result = []# 存每个页面里的多张图四元组信息
|
||||
items = page.get_images()
|
||||
dedup = set()
|
||||
for img in items:
|
||||
# 这里返回的是图片在page上的实际展示的大小。返回一个数组,每个元素第一部分是
|
||||
img_bojid = img[0]# 在pdf文件中是全局唯一的,如果这个图反复出现在pdf里那么就可能是垃圾信息,例如水印、页眉页脚等
|
||||
if img_bojid in junk_img_bojids:# 如果是垃圾图像,就跳过
|
||||
continue
|
||||
recs = page.get_image_rects(img, transform=True)
|
||||
if recs:
|
||||
rec = recs[0][0]
|
||||
x0, y0, x1, y1 = map(int, rec)
|
||||
width = x1 - x0
|
||||
height = y1 - y0
|
||||
if (x0, y0, x1, y1, img_bojid) in dedup: # 这里面会出现一些重复的bbox,无需重复出现,需要去掉
|
||||
continue
|
||||
if not all([width, height]): # 长和宽任何一个都不能是0,否则这个图片不可见,没有实际意义
|
||||
continue
|
||||
dedup.add((x0, y0, x1, y1, img_bojid))
|
||||
page_result.append([x0, y0, x1, y1, img_bojid])
|
||||
return page_result
|
||||
def get_image_info(doc: fitz.Document, page_width_pts, page_height_pts) -> list:
|
||||
"""
|
||||
返回每个页面里的图片的四元组,每个页面多个图片。
|
||||
:param doc:
|
||||
:return:
|
||||
"""
|
||||
# 使用 Counter 计数 img_bojid 的出现次数
|
||||
img_bojid_counter = Counter(img[0] for page in doc for img in page.get_images())
|
||||
# 找出出现次数超过 len(doc) 半数的 img_bojid
|
||||
|
||||
junk_limit = max(len(doc)*0.5, junk_limit_min)# 对一些页数比较少的进行豁免
|
||||
|
||||
junk_img_bojids = [img_bojid for img_bojid, count in img_bojid_counter.items() if count >= junk_limit]
|
||||
|
||||
#todo 加个判断,用前十页就行,这些垃圾图片需要满足两个条件,不止出现的次数要足够多,而且图片占书页面积的比例要足够大,且图与图大小都差不多
|
||||
#有两种扫描版,一种文字版,这里可能会有误判
|
||||
#扫描版1:每页都有所有扫描页图片,特点是图占比大,每页展示1张
|
||||
#扫描版2,每页存储的扫描页图片数量递增,特点是图占比大,每页展示1张,需要清空junklist跑前50页图片信息用于分类判断
|
||||
#文字版1.每页存储所有图片,特点是图片占页面比例不大,每页展示可能为0也可能不止1张 这种pdf需要拿前10页抽样检测img大小和个数,如果符合需要清空junklist
|
||||
imgs_len_list = [len(page.get_images()) for page in doc]
|
||||
|
||||
special_limit_pages = 10
|
||||
|
||||
# 统一用前十页结果做判断
|
||||
result = []
|
||||
break_loop = False
|
||||
for i, page in enumerate(doc):
|
||||
if break_loop:
|
||||
break
|
||||
if i >= special_limit_pages:
|
||||
break
|
||||
page_result = process_image(page) # 这里不传junk_img_bojids,拿前十页所有图片信息用于后续分析
|
||||
result.append(page_result)
|
||||
for item in result:
|
||||
if not any(item): # 如果任何一页没有图片,说明是个文字版,需要判断是否为特殊文字版
|
||||
if max(imgs_len_list) == min(imgs_len_list) and max(imgs_len_list) >= junk_limit_min:# 如果是特殊文字版,就把junklist置空并break
|
||||
junk_img_bojids = []
|
||||
else:# 不是特殊文字版,是个普通文字版,但是存在垃圾图片,不置空junklist
|
||||
pass
|
||||
break_loop = True
|
||||
break
|
||||
if not break_loop:
|
||||
# 获取前80%的元素
|
||||
top_eighty_percent = get_top_percent_list(imgs_len_list, 0.8)
|
||||
# 检查前80%的元素是否都相等
|
||||
if len(set(top_eighty_percent)) == 1 and max(imgs_len_list) >= junk_limit_min:
|
||||
|
||||
# # 如果前10页跑完都有图,根据每页图片数量是否相等判断是否需要清除junklist
|
||||
# if max(imgs_len_list) == min(imgs_len_list) and max(imgs_len_list) >= junk_limit_min:
|
||||
|
||||
#前10页都有图,且每页数量一致,需要检测图片大小占页面的比例判断是否需要清除junklist
|
||||
max_image_area_per_page = calculate_max_image_area_per_page(result, page_width_pts, page_height_pts)
|
||||
if len(max_image_area_per_page) < 0.8 * special_limit_pages: # 前10页不全是大图,说明可能是个文字版pdf,把垃圾图片list置空
|
||||
junk_img_bojids = []
|
||||
else:# 前10页都有图,而且80%都是大图,且每页图片数量一致并都很多,说明是扫描版1,不需要清空junklist
|
||||
pass
|
||||
else:# 每页图片数量不一致,需要清掉junklist全量跑前50页图片
|
||||
junk_img_bojids = []
|
||||
|
||||
#正式进入取前50页图片的信息流程
|
||||
result = []
|
||||
for i, page in enumerate(doc):
|
||||
if i >= scan_max_page:
|
||||
break
|
||||
page_result = process_image(page, junk_img_bojids)
|
||||
# logger.info(f"page {i} img_len: {len(page_result)}")
|
||||
result.append(page_result)
|
||||
|
||||
return result, junk_img_bojids
|
||||
|
||||
|
||||
def get_pdf_page_size_pts(doc: fitz.Document):
|
||||
page_cnt = len(doc)
|
||||
l: int = min(page_cnt, 50)
|
||||
#把所有宽度和高度塞到两个list 分别取中位数(中间遇到了个在纵页里塞横页的pdf,导致宽高互换了)
|
||||
page_width_list = []
|
||||
page_height_list = []
|
||||
for i in range(l):
|
||||
page = doc[i]
|
||||
page_rect = page.rect
|
||||
page_width_list.append(page_rect.width)
|
||||
page_height_list.append(page_rect.height)
|
||||
|
||||
page_width_list.sort()
|
||||
page_height_list.sort()
|
||||
|
||||
median_width = page_width_list[len(page_width_list) // 2]
|
||||
median_height = page_height_list[len(page_height_list) // 2]
|
||||
|
||||
|
||||
return median_width, median_height
|
||||
|
||||
|
||||
def get_pdf_textlen_per_page(doc: fitz.Document):
|
||||
text_len_lst = []
|
||||
for page in doc:
|
||||
# 拿包含img和text的所有blocks
|
||||
# text_block = page.get_text("blocks")
|
||||
# 拿所有text的blocks
|
||||
# text_block = page.get_text("words")
|
||||
# text_block_len = sum([len(t[4]) for t in text_block])
|
||||
#拿所有text的str
|
||||
text_block = page.get_text("text")
|
||||
text_block_len = len(text_block)
|
||||
# logger.info(f"page {page.number} text_block_len: {text_block_len}")
|
||||
text_len_lst.append(text_block_len)
|
||||
|
||||
return text_len_lst
|
||||
|
||||
def get_pdf_text_layout_per_page(doc: fitz.Document):
|
||||
"""
|
||||
根据PDF文档的每一页文本布局,判断该页的文本布局是横向、纵向还是未知。
|
||||
|
||||
Args:
|
||||
doc (fitz.Document): PDF文档对象。
|
||||
|
||||
Returns:
|
||||
List[str]: 每一页的文本布局(横向、纵向、未知)。
|
||||
|
||||
"""
|
||||
text_layout_list = []
|
||||
|
||||
for page_id, page in enumerate(doc):
|
||||
if page_id >= scan_max_page:
|
||||
break
|
||||
# 创建每一页的纵向和横向的文本行数计数器
|
||||
vertical_count = 0
|
||||
horizontal_count = 0
|
||||
text_dict = page.get_text("dict")
|
||||
if "blocks" in text_dict:
|
||||
for block in text_dict["blocks"]:
|
||||
if 'lines' in block:
|
||||
for line in block["lines"]:
|
||||
# 获取line的bbox顶点坐标
|
||||
x0, y0, x1, y1 = line['bbox']
|
||||
# 计算bbox的宽高
|
||||
width = x1 - x0
|
||||
height = y1 - y0
|
||||
# 计算bbox的面积
|
||||
area = width * height
|
||||
font_sizes = []
|
||||
for span in line['spans']:
|
||||
if 'size' in span:
|
||||
font_sizes.append(span['size'])
|
||||
if len(font_sizes) > 0:
|
||||
average_font_size = sum(font_sizes) / len(font_sizes)
|
||||
else:
|
||||
average_font_size = 10 # 有的line拿不到font_size,先定一个阈值100
|
||||
if area <= average_font_size ** 2: # 判断bbox的面积是否小于平均字体大小的平方,单字无法计算是横向还是纵向
|
||||
continue
|
||||
else:
|
||||
if 'wmode' in line: # 通过wmode判断文本方向
|
||||
if line['wmode'] == 1: # 判断是否为竖向文本
|
||||
vertical_count += 1
|
||||
elif line['wmode'] == 0: # 判断是否为横向文本
|
||||
horizontal_count += 1
|
||||
# if 'dir' in line: # 通过旋转角度计算判断文本方向
|
||||
# # 获取行的 "dir" 值
|
||||
# dir_value = line['dir']
|
||||
# cosine, sine = dir_value
|
||||
# # 计算角度
|
||||
# angle = math.degrees(math.acos(cosine))
|
||||
#
|
||||
# # 判断是否为横向文本
|
||||
# if abs(angle - 0) < 0.01 or abs(angle - 180) < 0.01:
|
||||
# # line_text = ' '.join(span['text'] for span in line['spans'])
|
||||
# # print('This line is horizontal:', line_text)
|
||||
# horizontal_count += 1
|
||||
# # 判断是否为纵向文本
|
||||
# elif abs(angle - 90) < 0.01 or abs(angle - 270) < 0.01:
|
||||
# # line_text = ' '.join(span['text'] for span in line['spans'])
|
||||
# # print('This line is vertical:', line_text)
|
||||
# vertical_count += 1
|
||||
# print(f"page_id: {page_id}, vertical_count: {vertical_count}, horizontal_count: {horizontal_count}")
|
||||
# 判断每一页的文本布局
|
||||
if vertical_count == 0 and horizontal_count == 0: # 该页没有文本,无法判断
|
||||
text_layout_list.append("unknow")
|
||||
continue
|
||||
else:
|
||||
if vertical_count > horizontal_count: # 该页的文本纵向行数大于横向的
|
||||
text_layout_list.append("vertical")
|
||||
else: # 该页的文本横向行数大于纵向的
|
||||
text_layout_list.append("horizontal")
|
||||
# logger.info(f"page_id: {page_id}, vertical_count: {vertical_count}, horizontal_count: {horizontal_count}")
|
||||
return text_layout_list
|
||||
|
||||
'''定义一个自定义异常用来抛出单页svg太多的pdf'''
|
||||
class PageSvgsTooManyError(Exception):
|
||||
def __init__(self, message="Page SVGs are too many"):
|
||||
self.message = message
|
||||
super().__init__(self.message)
|
||||
def get_svgs_per_page(doc: fitz.Document):
|
||||
svgs_len_list = []
|
||||
for page_id, page in enumerate(doc):
|
||||
# svgs = page.get_drawings()
|
||||
svgs = page.get_cdrawings() # 切换成get_cdrawings,效率更高
|
||||
len_svgs = len(svgs)
|
||||
if len_svgs >= 3000:
|
||||
raise PageSvgsTooManyError()
|
||||
else:
|
||||
svgs_len_list.append(len_svgs)
|
||||
# logger.info(f"page_id: {page_id}, svgs_len: {len(svgs)}")
|
||||
return svgs_len_list
|
||||
|
||||
def get_imgs_per_page(doc: fitz.Document):
|
||||
imgs_len_list = []
|
||||
for page_id, page in enumerate(doc):
|
||||
imgs = page.get_images()
|
||||
imgs_len_list.append(len(imgs))
|
||||
# logger.info(f"page_id: {page}, imgs_len: {len(imgs)}")
|
||||
|
||||
return imgs_len_list
|
||||
|
||||
|
||||
def get_language(doc: fitz.Document):
|
||||
"""
|
||||
获取PDF文档的语言。
|
||||
Args:
|
||||
doc (fitz.Document): PDF文档对象。
|
||||
Returns:
|
||||
str: 文档语言,如 "en-US"。
|
||||
"""
|
||||
language_lst = []
|
||||
for page_id, page in enumerate(doc):
|
||||
if page_id >= scan_max_page:
|
||||
break
|
||||
# 拿所有text的str
|
||||
text_block = page.get_text("text")
|
||||
page_language = detect_lang(text_block)
|
||||
language_lst.append(page_language)
|
||||
|
||||
# logger.info(f"page_id: {page_id}, page_language: {page_language}")
|
||||
|
||||
# 统计text_language_list中每种语言的个数
|
||||
count_dict = Counter(language_lst)
|
||||
# 输出text_language_list中出现的次数最多的语言
|
||||
language = max(count_dict, key=count_dict.get)
|
||||
return language
|
||||
|
||||
|
||||
def pdf_meta_scan(s3_pdf_path: str, pdf_bytes: bytes):
|
||||
"""
|
||||
:param s3_pdf_path:
|
||||
:param pdf_bytes: pdf文件的二进制数据
|
||||
几个维度来评价:是否加密,是否需要密码,纸张大小,总页数,是否文字可提取
|
||||
"""
|
||||
doc = fitz.open("pdf", pdf_bytes)
|
||||
is_needs_password = doc.needs_pass
|
||||
is_encrypted = doc.is_encrypted
|
||||
total_page = len(doc)
|
||||
if total_page == 0:
|
||||
logger.warning(f"drop this pdf: {s3_pdf_path}, drop_reason: {DropReason.EMPTY_PDF}")
|
||||
result = {"need_drop": True, "drop_reason": DropReason.EMPTY_PDF}
|
||||
return result
|
||||
else:
|
||||
page_width_pts, page_height_pts = get_pdf_page_size_pts(doc)
|
||||
# logger.info(f"page_width_pts: {page_width_pts}, page_height_pts: {page_height_pts}")
|
||||
|
||||
svgs_per_page = get_svgs_per_page(doc)
|
||||
# logger.info(f"svgs_per_page: {svgs_per_page}")
|
||||
imgs_per_page = get_imgs_per_page(doc)
|
||||
# logger.info(f"imgs_per_page: {imgs_per_page}")
|
||||
|
||||
image_info_per_page, junk_img_bojids = get_image_info(doc, page_width_pts, page_height_pts)
|
||||
# logger.info(f"image_info_per_page: {image_info_per_page}, junk_img_bojids: {junk_img_bojids}")
|
||||
text_len_per_page = get_pdf_textlen_per_page(doc)
|
||||
# logger.info(f"text_len_per_page: {text_len_per_page}")
|
||||
text_layout_per_page = get_pdf_text_layout_per_page(doc)
|
||||
# logger.info(f"text_layout_per_page: {text_layout_per_page}")
|
||||
text_language = get_language(doc)
|
||||
# logger.info(f"text_language: {text_language}")
|
||||
|
||||
|
||||
# 最后输出一条json
|
||||
res = {
|
||||
"pdf_path": s3_pdf_path,
|
||||
"is_needs_password": is_needs_password,
|
||||
"is_encrypted": is_encrypted,
|
||||
"total_page": total_page,
|
||||
"page_width_pts": int(page_width_pts),
|
||||
"page_height_pts": int(page_height_pts),
|
||||
"image_info_per_page": image_info_per_page,
|
||||
"text_len_per_page": text_len_per_page,
|
||||
"text_layout_per_page": text_layout_per_page,
|
||||
"text_language": text_language,
|
||||
"svgs_per_page": svgs_per_page,
|
||||
"imgs_per_page": imgs_per_page, # 增加每页img数量list
|
||||
"junk_img_bojids": junk_img_bojids, # 增加垃圾图片的bojid list
|
||||
"metadata": doc.metadata
|
||||
}
|
||||
# logger.info(json.dumps(res, ensure_ascii=False))
|
||||
return res
|
||||
|
||||
|
||||
@click.command()
|
||||
@click.option('--s3-pdf-path', help='s3上pdf文件的路径')
|
||||
@click.option('--s3-profile', help='s3上的profile')
|
||||
def main(s3_pdf_path: str, s3_profile: str):
|
||||
"""
|
||||
|
||||
"""
|
||||
try:
|
||||
file_content = read_file(s3_pdf_path, s3_profile)
|
||||
pdf_meta_scan(s3_pdf_path, file_content)
|
||||
except Exception as e:
|
||||
print(f"ERROR: {s3_pdf_path}, {e}", file=sys.stderr)
|
||||
logger.exception(e)
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
# "D:\project/20231108code-clean\pdf_cost_time\竖排例子\净空法师-大乘无量寿.pdf"
|
||||
# "D:\project/20231108code-clean\pdf_cost_time\竖排例子\三国演义_繁体竖排版.pdf"
|
||||
# "D:\project/20231108code-clean\pdf_cost_time\scihub\scihub_86800000\libgen.scimag86880000-86880999.zip_10.1021/acsami.1c03109.s002.pdf"
|
||||
# "D:/project/20231108code-clean/pdf_cost_time/scihub/scihub_18600000/libgen.scimag18645000-18645999.zip_10.1021/om3006239.pdf"
|
||||
# file_content = read_file("D:/project/20231108code-clean/pdf_cost_time/scihub/scihub_31000000/libgen.scimag31098000-31098999.zip_10.1109/isit.2006.261791.pdf","")
|
||||
# file_content = read_file("D:\project/20231108code-clean\pdf_cost_time\竖排例子\净空法师_大乘无量寿.pdf","")
|
||||
# doc = fitz.open("pdf", file_content)
|
||||
# text_layout_lst = get_pdf_text_layout_per_page(doc)
|
||||
# print(text_layout_lst)
|
||||
@@ -0,0 +1,681 @@
|
||||
# 定义这里的bbox是一个list [x0, y0, x1, y1, block_content, idx_x, idx_y, content_type, ext_x0, ext_y0, ext_x1, ext_y1], 初始时候idx_x, idx_y都是None
|
||||
# 其中x0, y0代表左上角坐标,x1, y1代表右下角坐标,坐标原点在左上角。
|
||||
|
||||
|
||||
|
||||
from layout.layout_spiler_recog import get_spilter_of_page
|
||||
from libs.boxbase import _is_bottom_full_overlap, _is_in, _is_in_or_part_overlap, _is_vertical_full_overlap
|
||||
from libs.commons import mymax
|
||||
|
||||
X0_IDX = 0
|
||||
Y0_IDX = 1
|
||||
X1_IDX = 2
|
||||
Y1_IDX = 3
|
||||
CONTENT_IDX = 4
|
||||
IDX_X = 5
|
||||
IDX_Y = 6
|
||||
CONTENT_TYPE_IDX = 7
|
||||
|
||||
X0_EXT_IDX = 8
|
||||
Y0_EXT_IDX = 9
|
||||
X1_EXT_IDX = 10
|
||||
Y1_EXT_IDX = 11
|
||||
|
||||
|
||||
def prepare_bboxes_for_layout_split(image_info, image_backup_info, table_info, inline_eq_info, interline_eq_info, text_raw_blocks: dict, page_boundry, page):
|
||||
"""
|
||||
text_raw_blocks:结构参考test/assets/papre/pymu_textblocks.json
|
||||
把bbox重新组装成一个list,每个元素[x0, y0, x1, y1, block_content, idx_x, idx_y, content_type, ext_x0, ext_y0, ext_x1, ext_y1], 初始时候idx_x, idx_y都是None. 对于图片、公式来说,block_content是图片的地址, 对于段落来说,block_content是pymupdf里的block结构
|
||||
"""
|
||||
all_bboxes = []
|
||||
|
||||
for image in image_info:
|
||||
box = image['bbox']
|
||||
# 由于没有实现横向的栏切分,因此在这里先过滤掉一些小的图片。这些图片有可能影响layout,造成没有横向栏切分的情况下,layout切分不准确。例如 scihub_76500000/libgen.scimag76570000-76570999.zip_10.1186/s13287-019-1355-1
|
||||
# 把长宽都小于50的去掉
|
||||
if abs(box[0]-box[2]) < 50 and abs(box[1]-box[3]) < 50:
|
||||
continue
|
||||
all_bboxes.append([box[0], box[1], box[2], box[3], None, None, None, 'image', None, None, None, None])
|
||||
|
||||
for table in table_info:
|
||||
box = table['bbox']
|
||||
all_bboxes.append([box[0], box[1], box[2], box[3], None, None, None, 'table', None, None, None, None])
|
||||
|
||||
"""由于公式与段落混合,因此公式不再参与layout划分,无需加入all_bboxes"""
|
||||
# 加入文本block
|
||||
text_block_temp = []
|
||||
for block in text_raw_blocks:
|
||||
bbox = block['bbox']
|
||||
text_block_temp.append([bbox[0], bbox[1], bbox[2], bbox[3], None, None, None, 'text', None, None, None, None])
|
||||
|
||||
text_block_new = resolve_bbox_overlap_for_layout_det(text_block_temp)
|
||||
text_block_new = filter_lines_bbox(text_block_new) # 去掉线条bbox,有可能让layout探测陷入无限循环
|
||||
|
||||
|
||||
"""找出会影响layout的色块、横向分割线"""
|
||||
spilter_bboxes = get_spilter_of_page(page, [b['bbox'] for b in image_info]+[b['bbox'] for b in image_backup_info], [b['bbox'] for b in table_info], )
|
||||
# 还要去掉存在于spilter_bboxes里的text_block
|
||||
if len(spilter_bboxes) > 0:
|
||||
text_block_new = [box for box in text_block_new if not any([_is_in_or_part_overlap(box[:4], spilter_bbox) for spilter_bbox in spilter_bboxes])]
|
||||
|
||||
for bbox in text_block_new:
|
||||
all_bboxes.append([bbox[0], bbox[1], bbox[2], bbox[3], None, None, None, 'text', None, None, None, None])
|
||||
|
||||
for bbox in spilter_bboxes:
|
||||
all_bboxes.append([bbox[0], bbox[1], bbox[2], bbox[3], None, None, None, 'spilter', None, None, None, None])
|
||||
|
||||
|
||||
return all_bboxes
|
||||
|
||||
def resolve_bbox_overlap_for_layout_det(bboxes:list):
|
||||
"""
|
||||
1. 去掉bbox互相包含的,去掉被包含的
|
||||
2. 上下方向上如果有重叠,就扩大大box范围,直到覆盖小box
|
||||
"""
|
||||
def _is_in_other_bbox(i:int):
|
||||
"""
|
||||
判断i个box是否被其他box有所包含
|
||||
"""
|
||||
for j in range(0, len(bboxes)):
|
||||
if j!=i and _is_in(bboxes[i][:4], bboxes[j][:4]):
|
||||
return True
|
||||
# elif j!=i and _is_bottom_full_overlap(bboxes[i][:4], bboxes[j][:4]):
|
||||
# return True
|
||||
|
||||
return False
|
||||
|
||||
# 首先去掉被包含的bbox
|
||||
new_bbox_1 = []
|
||||
for i in range(0, len(bboxes)):
|
||||
if not _is_in_other_bbox(i):
|
||||
new_bbox_1.append(bboxes[i])
|
||||
|
||||
# 其次扩展大的box
|
||||
new_box = []
|
||||
new_bbox_2 = []
|
||||
len_1 = len(new_bbox_2)
|
||||
while True:
|
||||
merged_idx = []
|
||||
for i in range(0, len(new_bbox_1)):
|
||||
if i in merged_idx:
|
||||
continue
|
||||
for j in range(i+1, len(new_bbox_1)):
|
||||
if j in merged_idx:
|
||||
continue
|
||||
bx1 = new_bbox_1[i]
|
||||
bx2 = new_bbox_1[j]
|
||||
if i!=j and _is_vertical_full_overlap(bx1[:4], bx2[:4]):
|
||||
merged_box = min([bx1[0], bx2[0]]), min([bx1[1], bx2[1]]), max([bx1[2], bx2[2]]), max([bx1[3], bx2[3]])
|
||||
new_bbox_2.append(merged_box)
|
||||
merged_idx.append(i)
|
||||
merged_idx.append(j)
|
||||
|
||||
for i in range(0, len(new_bbox_1)): # 没有合并的加入进来
|
||||
if i not in merged_idx:
|
||||
new_bbox_2.append(new_bbox_1[i])
|
||||
|
||||
if len(new_bbox_2)==0 or len_1==len(new_bbox_2):
|
||||
break
|
||||
else:
|
||||
len_1 = len(new_bbox_2)
|
||||
new_box = new_bbox_2
|
||||
new_bbox_1, new_bbox_2 = new_bbox_2, []
|
||||
|
||||
return new_box
|
||||
|
||||
|
||||
def filter_lines_bbox(bboxes: list):
|
||||
"""
|
||||
过滤掉bbox为空的行
|
||||
"""
|
||||
new_box = []
|
||||
for box in bboxes:
|
||||
x0, y0, x1, y1 = box[0], box[1], box[2], box[3]
|
||||
if abs(x0-x1)<=1 or abs(y0-y1)<=1:
|
||||
continue
|
||||
else:
|
||||
new_box.append(box)
|
||||
return new_box
|
||||
|
||||
|
||||
################################################################################
|
||||
# 第一种排序算法
|
||||
# 以下是基于延长线遮挡做的一个算法
|
||||
#
|
||||
################################################################################
|
||||
def find_all_left_bbox(this_bbox, all_bboxes) -> list:
|
||||
"""
|
||||
寻找this_bbox左边的所有bbox
|
||||
"""
|
||||
left_boxes = [box for box in all_bboxes if box[X1_IDX] <= this_bbox[X0_IDX]]
|
||||
return left_boxes
|
||||
|
||||
|
||||
def find_all_top_bbox(this_bbox, all_bboxes) -> list:
|
||||
"""
|
||||
寻找this_bbox上面的所有bbox
|
||||
"""
|
||||
top_boxes = [box for box in all_bboxes if box[Y1_IDX] <= this_bbox[Y0_IDX]]
|
||||
return top_boxes
|
||||
|
||||
|
||||
def get_and_set_idx_x(this_bbox, all_bboxes) -> int:
|
||||
"""
|
||||
寻找this_bbox在all_bboxes中的遮挡深度 idx_x
|
||||
"""
|
||||
if this_bbox[IDX_X] is not None:
|
||||
return this_bbox[IDX_X]
|
||||
else:
|
||||
all_left_bboxes = find_all_left_bbox(this_bbox, all_bboxes)
|
||||
if len(all_left_bboxes) == 0:
|
||||
this_bbox[IDX_X] = 0
|
||||
else:
|
||||
all_left_bboxes_idx = [get_and_set_idx_x(bbox, all_bboxes) for bbox in all_left_bboxes]
|
||||
max_idx_x = mymax(all_left_bboxes_idx)
|
||||
this_bbox[IDX_X] = max_idx_x + 1
|
||||
return this_bbox[IDX_X]
|
||||
|
||||
|
||||
def get_and_set_idx_y(this_bbox, all_bboxes) -> int:
|
||||
"""
|
||||
寻找this_bbox在all_bboxes中y方向的遮挡深度 idx_y
|
||||
"""
|
||||
if this_bbox[IDX_Y] is not None:
|
||||
return this_bbox[IDX_Y]
|
||||
else:
|
||||
all_top_bboxes = find_all_top_bbox(this_bbox, all_bboxes)
|
||||
if len(all_top_bboxes) == 0:
|
||||
this_bbox[IDX_Y] = 0
|
||||
else:
|
||||
all_top_bboxes_idx = [get_and_set_idx_y(bbox, all_bboxes) for bbox in all_top_bboxes]
|
||||
max_idx_y = mymax(all_top_bboxes_idx)
|
||||
this_bbox[IDX_Y] = max_idx_y + 1
|
||||
return this_bbox[IDX_Y]
|
||||
|
||||
|
||||
def bbox_sort(all_bboxes: list):
|
||||
"""
|
||||
排序
|
||||
"""
|
||||
all_bboxes_idx_x = [get_and_set_idx_x(bbox, all_bboxes) for bbox in all_bboxes]
|
||||
all_bboxes_idx_y = [get_and_set_idx_y(bbox, all_bboxes) for bbox in all_bboxes]
|
||||
all_bboxes_idx = [(idx_x, idx_y) for idx_x, idx_y in zip(all_bboxes_idx_x, all_bboxes_idx_y)]
|
||||
|
||||
all_bboxes_idx = [idx_x_y[0] * 100000 + idx_x_y[1] for idx_x_y in all_bboxes_idx] # 变换成一个点,保证能够先X,X相同时按Y排序
|
||||
all_bboxes_idx = list(zip(all_bboxes_idx, all_bboxes))
|
||||
all_bboxes_idx.sort(key=lambda x: x[0])
|
||||
sorted_bboxes = [bbox for idx, bbox in all_bboxes_idx]
|
||||
return sorted_bboxes
|
||||
|
||||
|
||||
################################################################################
|
||||
# 第二种排序算法
|
||||
# 下面的算法在计算idx_x和idx_y的时候不考虑延长线,而只考虑实际的长或者宽被遮挡的情况
|
||||
#
|
||||
################################################################################
|
||||
|
||||
def find_left_nearest_bbox(this_bbox, all_bboxes) -> list:
|
||||
"""
|
||||
在all_bboxes里找到所有右侧高度和this_bbox有重叠的bbox
|
||||
"""
|
||||
left_boxes = [box for box in all_bboxes if box[X1_IDX] <= this_bbox[X0_IDX] and any([
|
||||
box[Y0_IDX] < this_bbox[Y0_IDX] < box[Y1_IDX], box[Y0_IDX] < this_bbox[Y1_IDX] < box[Y1_IDX],
|
||||
this_bbox[Y0_IDX] < box[Y0_IDX] < this_bbox[Y1_IDX], this_bbox[Y0_IDX] < box[Y1_IDX] < this_bbox[Y1_IDX],
|
||||
box[Y0_IDX]==this_bbox[Y0_IDX] and box[Y1_IDX]==this_bbox[Y1_IDX]])]
|
||||
|
||||
# 然后再过滤一下,找到水平上距离this_bbox最近的那个
|
||||
if len(left_boxes) > 0:
|
||||
left_boxes.sort(key=lambda x: x[X1_IDX], reverse=True)
|
||||
left_boxes = [left_boxes[0]]
|
||||
else:
|
||||
left_boxes = []
|
||||
return left_boxes
|
||||
|
||||
|
||||
def get_and_set_idx_x_2(this_bbox, all_bboxes):
|
||||
"""
|
||||
寻找this_bbox在all_bboxes中的被直接遮挡的深度 idx_x
|
||||
这个遮挡深度不考虑延长线,而是被实际的长或者宽遮挡的情况
|
||||
"""
|
||||
if this_bbox[IDX_X] is not None:
|
||||
return this_bbox[IDX_X]
|
||||
else:
|
||||
left_nearest_bbox = find_left_nearest_bbox(this_bbox, all_bboxes)
|
||||
if len(left_nearest_bbox) == 0:
|
||||
this_bbox[IDX_X] = 0
|
||||
else:
|
||||
left_idx_x = get_and_set_idx_x_2(left_nearest_bbox[0], all_bboxes)
|
||||
this_bbox[IDX_X] = left_idx_x + 1
|
||||
return this_bbox[IDX_X]
|
||||
|
||||
|
||||
def find_top_nearest_bbox(this_bbox, all_bboxes) -> list:
|
||||
"""
|
||||
在all_bboxes里找到所有下侧宽度和this_bbox有重叠的bbox
|
||||
"""
|
||||
top_boxes = [box for box in all_bboxes if box[Y1_IDX] <= this_bbox[Y0_IDX] and any([
|
||||
box[X0_IDX] < this_bbox[X0_IDX] < box[X1_IDX], box[X0_IDX] < this_bbox[X1_IDX] < box[X1_IDX],
|
||||
this_bbox[X0_IDX] < box[X0_IDX] < this_bbox[X1_IDX], this_bbox[X0_IDX] < box[X1_IDX] < this_bbox[X1_IDX],
|
||||
box[X0_IDX]==this_bbox[X0_IDX] and box[X1_IDX]==this_bbox[X1_IDX]])]
|
||||
# 然后再过滤一下,找到水平上距离this_bbox最近的那个
|
||||
if len(top_boxes) > 0:
|
||||
top_boxes.sort(key=lambda x: x[Y1_IDX], reverse=True)
|
||||
top_boxes = [top_boxes[0]]
|
||||
else:
|
||||
top_boxes = []
|
||||
return top_boxes
|
||||
|
||||
|
||||
def get_and_set_idx_y_2(this_bbox, all_bboxes):
|
||||
"""
|
||||
寻找this_bbox在all_bboxes中的被直接遮挡的深度 idx_y
|
||||
这个遮挡深度不考虑延长线,而是被实际的长或者宽遮挡的情况
|
||||
"""
|
||||
if this_bbox[IDX_Y] is not None:
|
||||
return this_bbox[IDX_Y]
|
||||
else:
|
||||
top_nearest_bbox = find_top_nearest_bbox(this_bbox, all_bboxes)
|
||||
if len(top_nearest_bbox) == 0:
|
||||
this_bbox[IDX_Y] = 0
|
||||
else:
|
||||
top_idx_y = get_and_set_idx_y_2(top_nearest_bbox[0], all_bboxes)
|
||||
this_bbox[IDX_Y] = top_idx_y + 1
|
||||
return this_bbox[IDX_Y]
|
||||
|
||||
|
||||
def paper_bbox_sort(all_bboxes: list, page_width, page_height):
|
||||
all_bboxes_idx_x = [get_and_set_idx_x_2(bbox, all_bboxes) for bbox in all_bboxes]
|
||||
all_bboxes_idx_y = [get_and_set_idx_y_2(bbox, all_bboxes) for bbox in all_bboxes]
|
||||
all_bboxes_idx = [(idx_x, idx_y) for idx_x, idx_y in zip(all_bboxes_idx_x, all_bboxes_idx_y)]
|
||||
|
||||
all_bboxes_idx = [idx_x_y[0] * 100000 + idx_x_y[1] for idx_x_y in all_bboxes_idx] # 变换成一个点,保证能够先X,X相同时按Y排序
|
||||
all_bboxes_idx = list(zip(all_bboxes_idx, all_bboxes))
|
||||
all_bboxes_idx.sort(key=lambda x: x[0])
|
||||
sorted_bboxes = [bbox for idx, bbox in all_bboxes_idx]
|
||||
return sorted_bboxes
|
||||
|
||||
################################################################################
|
||||
"""
|
||||
第三种排序算法, 假设page的最左侧为X0,最右侧为X1,最上侧为Y0,最下侧为Y1
|
||||
这个排序算法在第二种算法基础上增加对bbox的预处理步骤。预处理思路如下:
|
||||
1. 首先在水平方向上对bbox进行扩展。扩展方法是:
|
||||
- 对每个bbox,找到其左边最近的bbox(也就是y方向有重叠),然后将其左边界扩展到左边最近bbox的右边界(x1+1),这里加1是为了避免重叠。如果没有左边的bbox,那么就将其左边界扩展到page的最左侧X0。
|
||||
- 对每个bbox,找到其右边最近的bbox(也就是y方向有重叠),然后将其右边界扩展到右边最近bbox的左边界(x0-1),这里减1是为了避免重叠。如果没有右边的bbox,那么就将其右边界扩展到page的最右侧X1。
|
||||
- 经过上面2个步骤,bbox扩展到了水平方向的最大范围。[左最近bbox.x1+1, 右最近bbox.x0-1]
|
||||
|
||||
2. 合并所有的连续水平方向的bbox, 合并方法是:
|
||||
- 对bbox进行y方向排序,然后从上到下遍历所有bbox,如果当前bbox和下一个bbox的x0, x1等于X0, X1,那么就合并这两个bbox。
|
||||
|
||||
3. 然后在垂直方向上对bbox进行扩展。扩展方法是:
|
||||
- 首先从page上切割掉合并后的水平bbox, 得到几个新的block
|
||||
针对每个block
|
||||
- x0: 扎到位于左侧x=x0延长线的左侧所有的bboxes, 找到最大的x1,让x0=x1+1。如果没有,则x0=X0
|
||||
- x1: 找到位于右侧x=x1延长线右侧所有的bboxes, 找到最小的x0, 让x1=x0-1。如果没有,则x1=X1
|
||||
随后在垂直方向上合并所有的连续的block,方法如下:
|
||||
- 对block进行x方向排序,然后从左到右遍历所有block,如果当前block和下一个block的x0, x1相等,那么就合并这两个block。
|
||||
如果垂直切分后所有小bbox都被分配到了一个block, 那么分割就完成了。这些合并后的block打上标签'GOOD_LAYOUT’
|
||||
如果在某个垂直方向上无法被完全分割到一个block,那么就将这个block打上标签'BAD_LAYOUT'。
|
||||
至此完成,一个页面的预处理,天然的block要么属于'GOOD_LAYOUT',要么属于'BAD_LAYOUT'。针对含有'BAD_LAYOUT'的页面,可以先按照自上而下,自左到右进行天然排序,也可以先过滤掉这种书籍。
|
||||
(完成条件下次加强:进行水平方向切分,把混乱的layout部分尽可能切割出去)
|
||||
"""
|
||||
################################################################################
|
||||
def find_left_neighbor_bboxes(this_bbox, all_bboxes) -> list:
|
||||
"""
|
||||
在all_bboxes里找到所有右侧高度和this_bbox有重叠的bbox
|
||||
这里使用扩展之后的bbox
|
||||
"""
|
||||
left_boxes = [box for box in all_bboxes if box[X1_EXT_IDX] <= this_bbox[X0_EXT_IDX] and any([
|
||||
box[Y0_EXT_IDX] < this_bbox[Y0_EXT_IDX] < box[Y1_EXT_IDX], box[Y0_EXT_IDX] < this_bbox[Y1_EXT_IDX] < box[Y1_EXT_IDX],
|
||||
this_bbox[Y0_EXT_IDX] < box[Y0_EXT_IDX] < this_bbox[Y1_EXT_IDX], this_bbox[Y0_EXT_IDX] < box[Y1_EXT_IDX] < this_bbox[Y1_EXT_IDX],
|
||||
box[Y0_EXT_IDX]==this_bbox[Y0_EXT_IDX] and box[Y1_EXT_IDX]==this_bbox[Y1_EXT_IDX]])]
|
||||
|
||||
# 然后再过滤一下,找到水平上距离this_bbox最近的那个
|
||||
if len(left_boxes) > 0:
|
||||
left_boxes.sort(key=lambda x: x[X1_EXT_IDX], reverse=True)
|
||||
left_boxes = left_boxes
|
||||
else:
|
||||
left_boxes = []
|
||||
return left_boxes
|
||||
|
||||
def find_top_neighbor_bboxes(this_bbox, all_bboxes) -> list:
|
||||
"""
|
||||
在all_bboxes里找到所有下侧宽度和this_bbox有重叠的bbox
|
||||
这里使用扩展之后的bbox
|
||||
"""
|
||||
top_boxes = [box for box in all_bboxes if box[Y1_EXT_IDX] <= this_bbox[Y0_EXT_IDX] and any([
|
||||
box[X0_EXT_IDX] < this_bbox[X0_EXT_IDX] < box[X1_EXT_IDX], box[X0_EXT_IDX] < this_bbox[X1_EXT_IDX] < box[X1_EXT_IDX],
|
||||
this_bbox[X0_EXT_IDX] < box[X0_EXT_IDX] < this_bbox[X1_EXT_IDX], this_bbox[X0_EXT_IDX] < box[X1_EXT_IDX] < this_bbox[X1_EXT_IDX],
|
||||
box[X0_EXT_IDX]==this_bbox[X0_EXT_IDX] and box[X1_EXT_IDX]==this_bbox[X1_EXT_IDX]])]
|
||||
# 然后再过滤一下,找到水平上距离this_bbox最近的那个
|
||||
if len(top_boxes) > 0:
|
||||
top_boxes.sort(key=lambda x: x[Y1_EXT_IDX], reverse=True)
|
||||
top_boxes = top_boxes
|
||||
else:
|
||||
top_boxes = []
|
||||
return top_boxes
|
||||
|
||||
def get_and_set_idx_x_2_ext(this_bbox, all_bboxes):
|
||||
"""
|
||||
寻找this_bbox在all_bboxes中的被直接遮挡的深度 idx_x
|
||||
这个遮挡深度不考虑延长线,而是被实际的长或者宽遮挡的情况
|
||||
"""
|
||||
if this_bbox[IDX_X] is not None:
|
||||
return this_bbox[IDX_X]
|
||||
else:
|
||||
left_nearest_bbox = find_left_neighbor_bboxes(this_bbox, all_bboxes)
|
||||
if len(left_nearest_bbox) == 0:
|
||||
this_bbox[IDX_X] = 0
|
||||
else:
|
||||
left_idx_x = [get_and_set_idx_x_2(b, all_bboxes) for b in left_nearest_bbox]
|
||||
this_bbox[IDX_X] = mymax(left_idx_x) + 1
|
||||
return this_bbox[IDX_X]
|
||||
|
||||
def get_and_set_idx_y_2_ext(this_bbox, all_bboxes):
|
||||
"""
|
||||
寻找this_bbox在all_bboxes中的被直接遮挡的深度 idx_y
|
||||
这个遮挡深度不考虑延长线,而是被实际的长或者宽遮挡的情况
|
||||
"""
|
||||
if this_bbox[IDX_Y] is not None:
|
||||
return this_bbox[IDX_Y]
|
||||
else:
|
||||
top_nearest_bbox = find_top_neighbor_bboxes(this_bbox, all_bboxes)
|
||||
if len(top_nearest_bbox) == 0:
|
||||
this_bbox[IDX_Y] = 0
|
||||
else:
|
||||
top_idx_y = [get_and_set_idx_y_2_ext(b, all_bboxes) for b in top_nearest_bbox]
|
||||
this_bbox[IDX_Y] = mymax(top_idx_y) + 1
|
||||
return this_bbox[IDX_Y]
|
||||
|
||||
def _paper_bbox_sort_ext(all_bboxes: list):
|
||||
all_bboxes_idx_x = [get_and_set_idx_x_2_ext(bbox, all_bboxes) for bbox in all_bboxes]
|
||||
all_bboxes_idx_y = [get_and_set_idx_y_2_ext(bbox, all_bboxes) for bbox in all_bboxes]
|
||||
all_bboxes_idx = [(idx_x, idx_y) for idx_x, idx_y in zip(all_bboxes_idx_x, all_bboxes_idx_y)]
|
||||
|
||||
all_bboxes_idx = [idx_x_y[0] * 100000 + idx_x_y[1] for idx_x_y in all_bboxes_idx] # 变换成一个点,保证能够先X,X相同时按Y排序
|
||||
all_bboxes_idx = list(zip(all_bboxes_idx, all_bboxes))
|
||||
all_bboxes_idx.sort(key=lambda x: x[0])
|
||||
sorted_bboxes = [bbox for idx, bbox in all_bboxes_idx]
|
||||
return sorted_bboxes
|
||||
|
||||
# ===============================================================================================
|
||||
def find_left_bbox_ext_line(this_bbox, all_bboxes) -> list:
|
||||
"""
|
||||
寻找this_bbox左边的所有bbox, 使用延长线
|
||||
"""
|
||||
left_boxes = [box for box in all_bboxes if box[X1_IDX] <= this_bbox[X0_IDX]]
|
||||
if len(left_boxes):
|
||||
left_boxes.sort(key=lambda x: x[X1_IDX], reverse=True)
|
||||
left_boxes = left_boxes[0]
|
||||
else:
|
||||
left_boxes = None
|
||||
|
||||
return left_boxes
|
||||
|
||||
def find_right_bbox_ext_line(this_bbox, all_bboxes) -> list:
|
||||
"""
|
||||
寻找this_bbox右边的所有bbox, 使用延长线
|
||||
"""
|
||||
right_boxes = [box for box in all_bboxes if box[X0_IDX] >= this_bbox[X1_IDX]]
|
||||
if len(right_boxes):
|
||||
right_boxes.sort(key=lambda x: x[X0_IDX])
|
||||
right_boxes = right_boxes[0]
|
||||
else:
|
||||
right_boxes = None
|
||||
return right_boxes
|
||||
|
||||
# =============================================================================================
|
||||
|
||||
def find_left_nearest_bbox_direct(this_bbox, all_bboxes) -> list:
|
||||
"""
|
||||
在all_bboxes里找到所有右侧高度和this_bbox有重叠的bbox, 不用延长线并且不能像
|
||||
"""
|
||||
left_boxes = [box for box in all_bboxes if box[X1_IDX] <= this_bbox[X0_IDX] and any([
|
||||
box[Y0_IDX] < this_bbox[Y0_IDX] < box[Y1_IDX], box[Y0_IDX] < this_bbox[Y1_IDX] < box[Y1_IDX],
|
||||
this_bbox[Y0_IDX] < box[Y0_IDX] < this_bbox[Y1_IDX], this_bbox[Y0_IDX] < box[Y1_IDX] < this_bbox[Y1_IDX],
|
||||
box[Y0_IDX]==this_bbox[Y0_IDX] and box[Y1_IDX]==this_bbox[Y1_IDX]])]
|
||||
|
||||
# 然后再过滤一下,找到水平上距离this_bbox最近的那个——x1最大的那个
|
||||
if len(left_boxes) > 0:
|
||||
left_boxes.sort(key=lambda x: x[X1_EXT_IDX] if x[X1_EXT_IDX] else x[X1_IDX], reverse=True)
|
||||
left_boxes = left_boxes[0]
|
||||
else:
|
||||
left_boxes = None
|
||||
return left_boxes
|
||||
|
||||
def find_right_nearst_bbox_direct(this_bbox, all_bboxes) -> list:
|
||||
"""
|
||||
找到在this_bbox右侧且距离this_bbox距离最近的bbox.必须是直接遮挡的那种
|
||||
"""
|
||||
right_bboxes = [box for box in all_bboxes if box[X0_IDX] >= this_bbox[X1_IDX] and any([
|
||||
this_bbox[Y0_IDX] < box[Y0_IDX] < this_bbox[Y1_IDX], this_bbox[Y0_IDX] < box[Y1_IDX] < this_bbox[Y1_IDX],
|
||||
box[Y0_IDX] < this_bbox[Y0_IDX] < box[Y1_IDX], box[Y0_IDX] < this_bbox[Y1_IDX] < box[Y1_IDX],
|
||||
box[Y0_IDX]==this_bbox[Y0_IDX] and box[Y1_IDX]==this_bbox[Y1_IDX]])]
|
||||
|
||||
if len(right_bboxes)>0:
|
||||
right_bboxes.sort(key=lambda x: x[X0_EXT_IDX] if x[X0_EXT_IDX] else x[X0_IDX])
|
||||
right_bboxes = right_bboxes[0]
|
||||
else:
|
||||
right_bboxes = None
|
||||
return right_bboxes
|
||||
|
||||
def reset_idx_x_y(all_boxes:list)->list:
|
||||
for box in all_boxes:
|
||||
box[IDX_X] = None
|
||||
box[IDX_Y] = None
|
||||
|
||||
return all_boxes
|
||||
|
||||
# ===================================================================================================
|
||||
def find_top_nearest_bbox_direct(this_bbox, bboxes_collection) -> list:
|
||||
"""
|
||||
找到在this_bbox上方且距离this_bbox距离最近的bbox.必须是直接遮挡的那种
|
||||
"""
|
||||
top_bboxes = [box for box in bboxes_collection if box[Y1_IDX] <= this_bbox[Y0_IDX] and any([
|
||||
box[X0_IDX] < this_bbox[X0_IDX] < box[X1_IDX], box[X0_IDX] < this_bbox[X1_IDX] < box[X1_IDX],
|
||||
this_bbox[X0_IDX] < box[X0_IDX] < this_bbox[X1_IDX], this_bbox[X0_IDX] < box[X1_IDX] < this_bbox[X1_IDX],
|
||||
box[X0_IDX]==this_bbox[X0_IDX] and box[X1_IDX]==this_bbox[X1_IDX]])]
|
||||
# 然后再过滤一下,找到上方距离this_bbox最近的那个
|
||||
if len(top_bboxes) > 0:
|
||||
top_bboxes.sort(key=lambda x: x[Y1_IDX], reverse=True)
|
||||
top_bboxes = top_bboxes[0]
|
||||
else:
|
||||
top_bboxes = None
|
||||
return top_bboxes
|
||||
|
||||
def find_bottom_nearest_bbox_direct(this_bbox, bboxes_collection) -> list:
|
||||
"""
|
||||
找到在this_bbox下方且距离this_bbox距离最近的bbox.必须是直接遮挡的那种
|
||||
"""
|
||||
bottom_bboxes = [box for box in bboxes_collection if box[Y0_IDX] >= this_bbox[Y1_IDX] and any([
|
||||
box[X0_IDX] < this_bbox[X0_IDX] < box[X1_IDX], box[X0_IDX] < this_bbox[X1_IDX] < box[X1_IDX],
|
||||
this_bbox[X0_IDX] < box[X0_IDX] < this_bbox[X1_IDX], this_bbox[X0_IDX] < box[X1_IDX] < this_bbox[X1_IDX],
|
||||
box[X0_IDX]==this_bbox[X0_IDX] and box[X1_IDX]==this_bbox[X1_IDX]])]
|
||||
# 然后再过滤一下,找到水平上距离this_bbox最近的那个
|
||||
if len(bottom_bboxes) > 0:
|
||||
bottom_bboxes.sort(key=lambda x: x[Y0_IDX])
|
||||
bottom_bboxes = bottom_bboxes[0]
|
||||
else:
|
||||
bottom_bboxes = None
|
||||
return bottom_bboxes
|
||||
|
||||
def find_boundry_bboxes(bboxes:list) -> tuple:
|
||||
"""
|
||||
找到bboxes的边界——找到所有bbox里最小的(x0, y0), 最大的(x1, y1)
|
||||
"""
|
||||
x0, y0, x1, y1 = bboxes[0][X0_IDX], bboxes[0][Y0_IDX], bboxes[0][X1_IDX], bboxes[0][Y1_IDX]
|
||||
for box in bboxes:
|
||||
x0 = min(box[X0_IDX], x0)
|
||||
y0 = min(box[Y0_IDX], y0)
|
||||
x1 = max(box[X1_IDX], x1)
|
||||
y1 = max(box[Y1_IDX], y1)
|
||||
|
||||
return x0, y0, x1, y1
|
||||
|
||||
|
||||
def extend_bbox_vertical(bboxes:list, boundry_x0, boundry_y0, boundry_x1, boundry_y1) -> list:
|
||||
"""
|
||||
在垂直方向上扩展能够直接垂直打通的bbox,也就是那些上下都没有其他box的bbox
|
||||
"""
|
||||
for box in bboxes:
|
||||
top_nearest_bbox = find_top_nearest_bbox_direct(box, bboxes)
|
||||
bottom_nearest_bbox = find_bottom_nearest_bbox_direct(box, bboxes)
|
||||
if top_nearest_bbox is None and bottom_nearest_bbox is None: # 独占一列
|
||||
box[X0_EXT_IDX] = box[X0_IDX]
|
||||
box[Y0_EXT_IDX] = boundry_y0
|
||||
box[X1_EXT_IDX] = box[X1_IDX]
|
||||
box[Y1_EXT_IDX] = boundry_y1
|
||||
# else:
|
||||
# if top_nearest_bbox is None:
|
||||
# box[Y0_EXT_IDX] = boundry_y0
|
||||
# else:
|
||||
# box[Y0_EXT_IDX] = top_nearest_bbox[Y1_IDX] + 1
|
||||
# if bottom_nearest_bbox is None:
|
||||
# box[Y1_EXT_IDX] = boundry_y1
|
||||
# else:
|
||||
# box[Y1_EXT_IDX] = bottom_nearest_bbox[Y0_IDX] - 1
|
||||
# box[X0_EXT_IDX] = box[X0_IDX]
|
||||
# box[X1_EXT_IDX] = box[X1_IDX]
|
||||
return bboxes
|
||||
|
||||
|
||||
# ===================================================================================================
|
||||
|
||||
def paper_bbox_sort_v2(all_bboxes: list, page_width:int, page_height:int):
|
||||
"""
|
||||
增加预处理行为的排序:
|
||||
return:
|
||||
[
|
||||
{
|
||||
"layout_bbox": [x0, y0, x1, y1],
|
||||
"layout_label":"GOOD_LAYOUT/BAD_LAYOUT",
|
||||
"content_bboxes": [] #每个元素都是[x0, y0, x1, y1, block_content, idx_x, idx_y, content_type, ext_x0, ext_y0, ext_x1, ext_y1], 并且顺序就是阅读顺序
|
||||
}
|
||||
]
|
||||
"""
|
||||
sorted_layouts = [] # 最后的返回结果
|
||||
page_x0, page_y0, page_x1, page_y1 = 1, 1, page_width-1, page_height-1
|
||||
|
||||
all_bboxes = paper_bbox_sort(all_bboxes) # 大致拍下序
|
||||
# 首先在水平方向上扩展独占一行的bbox
|
||||
for bbox in all_bboxes:
|
||||
left_nearest_bbox = find_left_nearest_bbox_direct(bbox, all_bboxes) # 非扩展线
|
||||
right_nearest_bbox = find_right_nearst_bbox_direct(bbox, all_bboxes)
|
||||
if left_nearest_bbox is None and right_nearest_bbox is None: # 独占一行
|
||||
bbox[X0_EXT_IDX] = page_x0
|
||||
bbox[Y0_EXT_IDX] = bbox[Y0_IDX]
|
||||
bbox[X1_EXT_IDX] = page_x1
|
||||
bbox[Y1_EXT_IDX] = bbox[Y1_IDX]
|
||||
|
||||
# 此时独占一行的被成功扩展到指定的边界上,这个时候利用边界条件合并连续的bbox,成为一个group
|
||||
if len(all_bboxes)==1:
|
||||
return [{"layout_bbox": [page_x0, page_y0, page_x1, page_y1], "layout_label":"GOOD_LAYOUT", "content_bboxes": all_bboxes}]
|
||||
if len(all_bboxes)==0:
|
||||
return []
|
||||
|
||||
"""
|
||||
然后合并所有连续水平方向的bbox.
|
||||
|
||||
"""
|
||||
all_bboxes.sort(key=lambda x: x[Y0_IDX])
|
||||
h_bboxes = []
|
||||
h_bbox_group = []
|
||||
v_boxes = []
|
||||
|
||||
for bbox in all_bboxes:
|
||||
if bbox[X0_IDX] == page_x0 and bbox[X1_IDX] == page_x1:
|
||||
h_bbox_group.append(bbox)
|
||||
else:
|
||||
if len(h_bbox_group)>0:
|
||||
h_bboxes.append(h_bbox_group)
|
||||
h_bbox_group = []
|
||||
# 最后一个group
|
||||
if len(h_bbox_group)>0:
|
||||
h_bboxes.append(h_bbox_group)
|
||||
|
||||
"""
|
||||
现在h_bboxes里面是所有的group了,每个group都是一个list
|
||||
对h_bboxes里的每个group进行计算放回到sorted_layouts里
|
||||
"""
|
||||
for gp in h_bboxes:
|
||||
gp.sort(key=lambda x: x[Y0_IDX])
|
||||
block_info = {"layout_label":"GOOD_LAYOUT", "content_bboxes": gp}
|
||||
# 然后计算这个group的layout_bbox,也就是最小的x0,y0, 最大的x1,y1
|
||||
x0, y0, x1, y1 = gp[0][X0_EXT_IDX], gp[0][Y0_EXT_IDX], gp[-1][X1_EXT_IDX], gp[-1][Y1_EXT_IDX]
|
||||
block_info["layout_bbox"] = [x0, y0, x1, y1]
|
||||
sorted_layouts.append(block_info)
|
||||
|
||||
# 接下来利用这些连续的水平bbox的layout_bbox的y0, y1,从水平上切分开其余的为几个部分
|
||||
h_split_lines = [page_y0]
|
||||
for gp in h_bboxes:
|
||||
layout_bbox = gp['layout_bbox']
|
||||
y0, y1 = layout_bbox[1], layout_bbox[3]
|
||||
h_split_lines.append(y0)
|
||||
h_split_lines.append(y1)
|
||||
h_split_lines.append(page_y1)
|
||||
|
||||
unsplited_bboxes = []
|
||||
for i in range(0, len(h_split_lines), 2):
|
||||
start_y0, start_y1 = h_split_lines[i:i+2]
|
||||
# 然后找出[start_y0, start_y1]之间的其他bbox,这些组成一个未分割板块
|
||||
bboxes_in_block = [bbox for bbox in all_bboxes if bbox[Y0_IDX]>=start_y0 and bbox[Y1_IDX]<=start_y1]
|
||||
unsplited_bboxes.append(bboxes_in_block)
|
||||
# ================== 至此,水平方向的 已经切分排序完毕====================================
|
||||
"""
|
||||
接下来针对每个非水平的部分切分垂直方向的
|
||||
此时,只剩下了无法被完全水平打通的bbox了。对这些box,优先进行垂直扩展,然后进行垂直切分.
|
||||
分3步:
|
||||
1. 先把能完全垂直打通的隔离出去当做一个layout
|
||||
2. 其余的先垂直切分
|
||||
3. 垂直切分之后的部分再尝试水平切分
|
||||
4. 剩下的不能被切分的各个部分当成一个layout
|
||||
"""
|
||||
# 对每部分进行垂直切分
|
||||
for bboxes_in_block in unsplited_bboxes:
|
||||
# 首先对这个block的bbox进行垂直方向上的扩展
|
||||
boundry_x0, boundry_y0, boundry_x1, boundry_y1 = find_boundry_bboxes(bboxes_in_block)
|
||||
# 进行垂直方向上的扩展
|
||||
extended_vertical_bboxes = extend_bbox_vertical(bboxes_in_block, boundry_x0, boundry_y0, boundry_x1, boundry_y1)
|
||||
# 然后对这个block进行垂直方向上的切分
|
||||
extend_bbox_vertical.sort(key=lambda x: x[X0_IDX]) # x方向上从小到大,代表了从左到右读取
|
||||
v_boxes_group = []
|
||||
for bbox in extended_vertical_bboxes:
|
||||
if bbox[Y0_IDX]==boundry_y0 and bbox[Y1_IDX]==boundry_y1:
|
||||
v_boxes_group.append(bbox)
|
||||
else:
|
||||
if len(v_boxes_group)>0:
|
||||
v_boxes.append(v_boxes_group)
|
||||
v_boxes_group = []
|
||||
|
||||
if len(v_boxes_group)>0:
|
||||
|
||||
v_boxes.append(v_boxes_group)
|
||||
|
||||
# 把连续的垂直部分加入到sorted_layouts里。注意这个时候已经是连续的垂直部分了,因为上面已经做了
|
||||
for gp in v_boxes:
|
||||
gp.sort(key=lambda x: x[X0_IDX])
|
||||
block_info = {"layout_label":"GOOD_LAYOUT", "content_bboxes": gp}
|
||||
# 然后计算这个group的layout_bbox,也就是最小的x0,y0, 最大的x1,y1
|
||||
x0, y0, x1, y1 = gp[0][X0_EXT_IDX], gp[0][Y0_EXT_IDX], gp[-1][X1_EXT_IDX], gp[-1][Y1_EXT_IDX]
|
||||
block_info["layout_bbox"] = [x0, y0, x1, y1]
|
||||
sorted_layouts.append(block_info)
|
||||
|
||||
# 在垂直方向上,划分子块,也就是用贯通的垂直线进行切分。这些被切分出来的块,极大可能是可被垂直切分的,如果不能完全的垂直切分,那么尝试水平切分。都不能的则当成一个layout
|
||||
v_split_lines = [boundry_x0]
|
||||
for gp in v_boxes:
|
||||
layout_bbox = gp['layout_bbox']
|
||||
x0, x1 = layout_bbox[0], layout_bbox[2]
|
||||
v_split_lines.append(x0)
|
||||
v_split_lines.append(x1)
|
||||
v_split_lines.append(boundry_x1)
|
||||
|
||||
reset_idx_x_y(all_bboxes)
|
||||
all_boxes = _paper_bbox_sort_ext(all_bboxes)
|
||||
return all_boxes
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,182 @@
|
||||
from layout.bbox_sort import X0_EXT_IDX, X0_IDX, X1_EXT_IDX, X1_IDX, Y0_EXT_IDX, Y0_IDX, Y1_EXT_IDX, Y1_IDX
|
||||
from libs.boxbase import _is_bottom_full_overlap, _left_intersect, _right_intersect
|
||||
|
||||
|
||||
def find_all_left_bbox_direct(this_bbox, all_bboxes) -> list:
|
||||
"""
|
||||
在all_bboxes里找到所有右侧垂直方向上和this_bbox有重叠的bbox, 不用延长线
|
||||
并且要考虑两个box左右相交的情况,如果相交了,那么右侧的box就不算最左侧。
|
||||
"""
|
||||
left_boxes = [box for box in all_bboxes if box[X1_IDX] <= this_bbox[X0_IDX]
|
||||
and any([
|
||||
box[Y0_IDX] < this_bbox[Y0_IDX] < box[Y1_IDX], box[Y0_IDX] < this_bbox[Y1_IDX] < box[Y1_IDX],
|
||||
this_bbox[Y0_IDX] < box[Y0_IDX] < this_bbox[Y1_IDX], this_bbox[Y0_IDX] < box[Y1_IDX] < this_bbox[Y1_IDX],
|
||||
box[Y0_IDX]==this_bbox[Y0_IDX] and box[Y1_IDX]==this_bbox[Y1_IDX]]) or _left_intersect(box[:4], this_bbox[:4])]
|
||||
|
||||
# 然后再过滤一下,找到水平上距离this_bbox最近的那个——x1最大的那个
|
||||
if len(left_boxes) > 0:
|
||||
left_boxes.sort(key=lambda x: x[X1_EXT_IDX] if x[X1_EXT_IDX] else x[X1_IDX], reverse=True)
|
||||
left_boxes = left_boxes[0]
|
||||
else:
|
||||
left_boxes = None
|
||||
return left_boxes
|
||||
|
||||
def find_all_right_bbox_direct(this_bbox, all_bboxes) -> list:
|
||||
"""
|
||||
找到在this_bbox右侧且距离this_bbox距离最近的bbox.必须是直接遮挡的那种
|
||||
"""
|
||||
right_bboxes = [box for box in all_bboxes if box[X0_IDX] >= this_bbox[X1_IDX]
|
||||
and any([
|
||||
this_bbox[Y0_IDX] < box[Y0_IDX] < this_bbox[Y1_IDX], this_bbox[Y0_IDX] < box[Y1_IDX] < this_bbox[Y1_IDX],
|
||||
box[Y0_IDX] < this_bbox[Y0_IDX] < box[Y1_IDX], box[Y0_IDX] < this_bbox[Y1_IDX] < box[Y1_IDX],
|
||||
box[Y0_IDX]==this_bbox[Y0_IDX] and box[Y1_IDX]==this_bbox[Y1_IDX]]) or _right_intersect(this_bbox[:4], box[:4])]
|
||||
|
||||
if len(right_bboxes)>0:
|
||||
right_bboxes.sort(key=lambda x: x[X0_EXT_IDX] if x[X0_EXT_IDX] else x[X0_IDX])
|
||||
right_bboxes = right_bboxes[0]
|
||||
else:
|
||||
right_bboxes = None
|
||||
return right_bboxes
|
||||
|
||||
def find_all_top_bbox_direct(this_bbox, all_bboxes) -> list:
|
||||
"""
|
||||
找到在this_bbox上侧且距离this_bbox距离最近的bbox.必须是直接遮挡的那种
|
||||
"""
|
||||
top_bboxes = [box for box in all_bboxes if box[Y1_IDX] <= this_bbox[Y0_IDX] and any([
|
||||
box[X0_IDX] < this_bbox[X0_IDX] < box[X1_IDX], box[X0_IDX] < this_bbox[X1_IDX] < box[X1_IDX],
|
||||
this_bbox[X0_IDX] < box[X0_IDX] < this_bbox[X1_IDX], this_bbox[X0_IDX] < box[X1_IDX] < this_bbox[X1_IDX],
|
||||
box[X0_IDX]==this_bbox[X0_IDX] and box[X1_IDX]==this_bbox[X1_IDX]])]
|
||||
|
||||
if len(top_bboxes)>0:
|
||||
top_bboxes.sort(key=lambda x: x[Y1_EXT_IDX] if x[Y1_EXT_IDX] else x[Y1_IDX], reverse=True)
|
||||
top_bboxes = top_bboxes[0]
|
||||
else:
|
||||
top_bboxes = None
|
||||
return top_bboxes
|
||||
|
||||
def find_all_bottom_bbox_direct(this_bbox, all_bboxes) -> list:
|
||||
"""
|
||||
找到在this_bbox下侧且距离this_bbox距离最近的bbox.必须是直接遮挡的那种
|
||||
"""
|
||||
bottom_bboxes = [box for box in all_bboxes if box[Y0_IDX] >= this_bbox[Y1_IDX] and any([
|
||||
this_bbox[X0_IDX] < box[X0_IDX] < this_bbox[X1_IDX], this_bbox[X0_IDX] < box[X1_IDX] < this_bbox[X1_IDX],
|
||||
box[X0_IDX] < this_bbox[X0_IDX] < box[X1_IDX], box[X0_IDX] < this_bbox[X1_IDX] < box[X1_IDX],
|
||||
box[X0_IDX]==this_bbox[X0_IDX] and box[X1_IDX]==this_bbox[X1_IDX]])]
|
||||
|
||||
if len(bottom_bboxes)>0:
|
||||
bottom_bboxes.sort(key=lambda x: x[Y0_IDX])
|
||||
bottom_bboxes = bottom_bboxes[0]
|
||||
else:
|
||||
bottom_bboxes = None
|
||||
return bottom_bboxes
|
||||
|
||||
# ===================================================================================================================
|
||||
def find_bottom_bbox_direct_from_right_edge(this_bbox, all_bboxes) -> list:
|
||||
"""
|
||||
找到在this_bbox下侧且距离this_bbox距离最近的bbox.必须是直接遮挡的那种
|
||||
"""
|
||||
bottom_bboxes = [box for box in all_bboxes if box[Y0_IDX] >= this_bbox[Y1_IDX] and any([
|
||||
this_bbox[X0_IDX] < box[X0_IDX] < this_bbox[X1_IDX], this_bbox[X0_IDX] < box[X1_IDX] < this_bbox[X1_IDX],
|
||||
box[X0_IDX] < this_bbox[X0_IDX] < box[X1_IDX], box[X0_IDX] < this_bbox[X1_IDX] < box[X1_IDX],
|
||||
box[X0_IDX]==this_bbox[X0_IDX] and box[X1_IDX]==this_bbox[X1_IDX]])]
|
||||
|
||||
if len(bottom_bboxes)>0:
|
||||
# y0最小, X1最大的那个,也就是box上边缘最靠近this_bbox的那个,并且还最靠右
|
||||
bottom_bboxes.sort(key=lambda x: x[Y0_IDX])
|
||||
bottom_bboxes = [box for box in bottom_bboxes if box[Y0_IDX]==bottom_bboxes[0][Y0_IDX]]
|
||||
# 然后再y1相同的情况下,找到x1最大的那个
|
||||
bottom_bboxes.sort(key=lambda x: x[X1_IDX], reverse=True)
|
||||
bottom_bboxes = bottom_bboxes[0]
|
||||
else:
|
||||
bottom_bboxes = None
|
||||
return bottom_bboxes
|
||||
|
||||
def find_bottom_bbox_direct_from_left_edge(this_bbox, all_bboxes) -> list:
|
||||
"""
|
||||
找到在this_bbox下侧且距离this_bbox距离最近的bbox.必须是直接遮挡的那种
|
||||
"""
|
||||
bottom_bboxes = [box for box in all_bboxes if box[Y0_IDX] >= this_bbox[Y1_IDX] and any([
|
||||
this_bbox[X0_IDX] < box[X0_IDX] < this_bbox[X1_IDX], this_bbox[X0_IDX] < box[X1_IDX] < this_bbox[X1_IDX],
|
||||
box[X0_IDX] < this_bbox[X0_IDX] < box[X1_IDX], box[X0_IDX] < this_bbox[X1_IDX] < box[X1_IDX],
|
||||
box[X0_IDX]==this_bbox[X0_IDX] and box[X1_IDX]==this_bbox[X1_IDX]])]
|
||||
|
||||
if len(bottom_bboxes)>0:
|
||||
# y0最小, X0最小的那个
|
||||
bottom_bboxes.sort(key=lambda x: x[Y0_IDX])
|
||||
bottom_bboxes = [box for box in bottom_bboxes if box[Y0_IDX]==bottom_bboxes[0][Y0_IDX]]
|
||||
# 然后再y0相同的情况下,找到x0最小的那个
|
||||
bottom_bboxes.sort(key=lambda x: x[X0_IDX])
|
||||
bottom_bboxes = bottom_bboxes[0]
|
||||
else:
|
||||
bottom_bboxes = None
|
||||
return bottom_bboxes
|
||||
|
||||
def find_top_bbox_direct_from_left_edge(this_bbox, all_bboxes) -> list:
|
||||
"""
|
||||
找到在this_bbox上侧且距离this_bbox距离最近的bbox.必须是直接遮挡的那种
|
||||
"""
|
||||
top_bboxes = [box for box in all_bboxes if box[Y1_IDX] <= this_bbox[Y0_IDX] and any([
|
||||
box[X0_IDX] < this_bbox[X0_IDX] < box[X1_IDX], box[X0_IDX] < this_bbox[X1_IDX] < box[X1_IDX],
|
||||
this_bbox[X0_IDX] < box[X0_IDX] < this_bbox[X1_IDX], this_bbox[X0_IDX] < box[X1_IDX] < this_bbox[X1_IDX],
|
||||
box[X0_IDX]==this_bbox[X0_IDX] and box[X1_IDX]==this_bbox[X1_IDX]])]
|
||||
|
||||
if len(top_bboxes)>0:
|
||||
# y1最大, X0最小的那个
|
||||
top_bboxes.sort(key=lambda x: x[Y1_IDX], reverse=True)
|
||||
top_bboxes = [box for box in top_bboxes if box[Y1_IDX]==top_bboxes[0][Y1_IDX]]
|
||||
# 然后再y1相同的情况下,找到x0最小的那个
|
||||
top_bboxes.sort(key=lambda x: x[X0_IDX])
|
||||
top_bboxes = top_bboxes[0]
|
||||
else:
|
||||
top_bboxes = None
|
||||
return top_bboxes
|
||||
|
||||
def find_top_bbox_direct_from_right_edge(this_bbox, all_bboxes) -> list:
|
||||
"""
|
||||
找到在this_bbox上侧且距离this_bbox距离最近的bbox.必须是直接遮挡的那种
|
||||
"""
|
||||
top_bboxes = [box for box in all_bboxes if box[Y1_IDX] <= this_bbox[Y0_IDX] and any([
|
||||
box[X0_IDX] < this_bbox[X0_IDX] < box[X1_IDX], box[X0_IDX] < this_bbox[X1_IDX] < box[X1_IDX],
|
||||
this_bbox[X0_IDX] < box[X0_IDX] < this_bbox[X1_IDX], this_bbox[X0_IDX] < box[X1_IDX] < this_bbox[X1_IDX],
|
||||
box[X0_IDX]==this_bbox[X0_IDX] and box[X1_IDX]==this_bbox[X1_IDX]])]
|
||||
|
||||
if len(top_bboxes)>0:
|
||||
# y1最大, X1最大的那个
|
||||
top_bboxes.sort(key=lambda x: x[Y1_IDX], reverse=True)
|
||||
top_bboxes = [box for box in top_bboxes if box[Y1_IDX]==top_bboxes[0][Y1_IDX]]
|
||||
# 然后再y1相同的情况下,找到x1最大的那个
|
||||
top_bboxes.sort(key=lambda x: x[X1_IDX], reverse=True)
|
||||
top_bboxes = top_bboxes[0]
|
||||
else:
|
||||
top_bboxes = None
|
||||
return top_bboxes
|
||||
|
||||
# ===================================================================================================================
|
||||
|
||||
def get_left_edge_bboxes(all_bboxes) -> list:
|
||||
"""
|
||||
返回最左边的bbox
|
||||
"""
|
||||
left_bboxes = [box for box in all_bboxes if find_all_left_bbox_direct(box, all_bboxes) is None]
|
||||
return left_bboxes
|
||||
|
||||
def get_right_edge_bboxes(all_bboxes) -> list:
|
||||
"""
|
||||
返回最右边的bbox
|
||||
"""
|
||||
right_bboxes = [box for box in all_bboxes if find_all_right_bbox_direct(box, all_bboxes) is None]
|
||||
return right_bboxes
|
||||
|
||||
def fix_vertical_bbox_pos(bboxes:list):
|
||||
"""
|
||||
检查这批bbox在垂直方向是否有轻微的重叠,如果重叠了,就把重叠的bbox往下移动一点
|
||||
在x方向上必须一个包含或者被包含,或者完全重叠,不能只有部分重叠
|
||||
"""
|
||||
bboxes.sort(key=lambda x: x[Y0_IDX]) # 从上向下排列
|
||||
for i in range(0, len(bboxes)):
|
||||
for j in range(i+1, len(bboxes)):
|
||||
if _is_bottom_full_overlap(bboxes[i][:4], bboxes[j][:4]):
|
||||
# 如果两个bbox有部分重叠,那么就把下面的bbox往下移动一点
|
||||
bboxes[j][Y0_IDX] = bboxes[i][Y1_IDX] + 2 # 2是个经验值
|
||||
break
|
||||
return bboxes
|
||||
@@ -0,0 +1,733 @@
|
||||
"""
|
||||
对pdf上的box进行layout识别,并对内部组成的box进行排序
|
||||
"""
|
||||
|
||||
import json
|
||||
from loguru import logger
|
||||
from layout.bbox_sort import CONTENT_IDX, CONTENT_TYPE_IDX, X0_EXT_IDX, X0_IDX, X1_EXT_IDX, X1_IDX, Y0_EXT_IDX, Y0_IDX, Y1_EXT_IDX, Y1_IDX, paper_bbox_sort
|
||||
from layout.layout_det_utils import find_all_left_bbox_direct, find_all_right_bbox_direct, find_bottom_bbox_direct_from_left_edge, find_bottom_bbox_direct_from_right_edge, find_top_bbox_direct_from_left_edge, find_top_bbox_direct_from_right_edge, find_all_top_bbox_direct, find_all_bottom_bbox_direct, get_left_edge_bboxes, get_right_edge_bboxes
|
||||
from libs.boxbase import get_bbox_in_boundry
|
||||
|
||||
|
||||
LAYOUT_V = "V"
|
||||
LAYOUT_H = "H"
|
||||
LAYOUT_UNPROC = "U"
|
||||
LAYOUT_BAD = "B"
|
||||
|
||||
def _is_single_line_text(bbox):
|
||||
"""
|
||||
检查bbox里面的文字是否只有一行
|
||||
"""
|
||||
return True # TODO
|
||||
box_type = bbox[CONTENT_TYPE_IDX]
|
||||
if box_type != 'text':
|
||||
return False
|
||||
paras = bbox[CONTENT_IDX]["paras"]
|
||||
text_content = ""
|
||||
for para_id, para in paras.items(): # 拼装内部的段落文本
|
||||
is_title = para['is_title']
|
||||
if is_title!=0:
|
||||
text_content += f"## {para['text']}"
|
||||
else:
|
||||
text_content += para["text"]
|
||||
text_content += "\n\n"
|
||||
|
||||
return bbox[CONTENT_TYPE_IDX] == 'text' and len(text_content.split("\n\n")) <= 1
|
||||
|
||||
|
||||
def _horizontal_split(bboxes:list, boundry:tuple, avg_font_size=20)-> list:
|
||||
"""
|
||||
对bboxes进行水平切割
|
||||
方法是:找到左侧和右侧都没有被直接遮挡的box,然后进行扩展,之后进行切割
|
||||
return:
|
||||
返回几个大的Layout区域 [[x0, y0, x1, y1, "h|u|v"], ], h代表水平,u代表未探测的,v代表垂直布局
|
||||
"""
|
||||
sorted_layout_blocks = [] # 这是要最终返回的值
|
||||
|
||||
bound_x0, bound_y0, bound_x1, bound_y1 = boundry
|
||||
all_bboxes = get_bbox_in_boundry(bboxes, boundry)
|
||||
#all_bboxes = paper_bbox_sort(all_bboxes, abs(bound_x1-bound_x0), abs(bound_y1-bound_x0)) # 大致拍下序, 这个是基于直接遮挡的。
|
||||
"""
|
||||
首先在水平方向上扩展独占一行的bbox
|
||||
|
||||
"""
|
||||
last_h_split_line_y1 = bound_y0 #记录下上次的水平分割线
|
||||
for i, bbox in enumerate(all_bboxes):
|
||||
left_nearest_bbox = find_all_left_bbox_direct(bbox, all_bboxes) # 非扩展线
|
||||
right_nearest_bbox = find_all_right_bbox_direct(bbox, all_bboxes)
|
||||
if left_nearest_bbox is None and right_nearest_bbox is None: # 独占一行
|
||||
"""
|
||||
然而,如果只是孤立的一行文字,那么就还要满足以下几个条件才可以:
|
||||
1. bbox和中心线相交。或者
|
||||
2. 上方或者下方也存在同类水平的独占一行的bbox。 或者
|
||||
3. TODO 加强条件:这个bbox上方和下方是同一列column,那么就不能算作独占一行
|
||||
"""
|
||||
# 先检查这个bbox里是否只包含一行文字
|
||||
is_single_line = _is_single_line_text(bbox)
|
||||
"""
|
||||
这里有个点需要注意,当页面内容不是居中的时候,第一次调用传递的是page的boundry,这个时候mid_x就不是中心线了.
|
||||
所以这里计算出最紧致的boundry,然后再计算mid_x
|
||||
"""
|
||||
boundry_real_x0, boundry_real_x1 = min([bbox[X0_IDX] for bbox in all_bboxes]), max([bbox[X1_IDX] for bbox in all_bboxes])
|
||||
mid_x = (boundry_real_x0+boundry_real_x1)/2
|
||||
# 检查这个box是否内容在中心线有交
|
||||
# 必须跨过去2个字符的宽度
|
||||
is_cross_boundry_mid_line = min(mid_x-bbox[X0_IDX], bbox[X1_IDX]-mid_x) > avg_font_size*2
|
||||
"""
|
||||
检查条件2
|
||||
"""
|
||||
is_belong_to_col = False
|
||||
"""
|
||||
检查是否能被上方col吸收,方法是:
|
||||
1. 上方非空且不是独占一行的,并且
|
||||
2. 从上个水平分割的最大y=y1开始到当前bbox,最左侧的bbox的[min_x0, max_x1],能够覆盖当前box的[x0, x1]
|
||||
"""
|
||||
"""
|
||||
以迭代的方式向上找,查找范围是[bound_x0, last_h_sp, bound_x1, bbox[Y0_IDX]]
|
||||
"""
|
||||
#先确定上方的y0, y0
|
||||
b_y0, b_y1 = last_h_split_line_y1, bbox[Y0_IDX]
|
||||
#然后从box开始逐个向上找到所有与box在x上有交集的box
|
||||
box_to_check = [bound_x0, b_y0, bound_x1, b_y1]
|
||||
bbox_in_bound_check = get_bbox_in_boundry(all_bboxes, box_to_check)
|
||||
|
||||
bboxes_on_top = []
|
||||
virtual_box = bbox
|
||||
while True:
|
||||
b_on_top = find_all_top_bbox_direct(virtual_box, bbox_in_bound_check)
|
||||
if b_on_top is not None:
|
||||
bboxes_on_top.append(b_on_top)
|
||||
virtual_box = [min([virtual_box[X0_IDX], b_on_top[X0_IDX]]), min(virtual_box[Y0_IDX], b_on_top[Y0_IDX]), max([virtual_box[X1_IDX], b_on_top[X1_IDX]]), b_y1]
|
||||
else:
|
||||
break
|
||||
|
||||
# 随后确定这些box的最小x0, 最大x1
|
||||
if len(bboxes_on_top)>0 and len(bboxes_on_top) != len(bbox_in_bound_check):# virtual_box可能会膨胀到占满整个区域,这实际上就不能属于一个col了。
|
||||
min_x0, max_x1 = virtual_box[X0_IDX], virtual_box[X1_IDX]
|
||||
# 然后采用一种比较粗糙的方法,看min_x0,max_x1是否与位于[bound_x0, last_h_sp, bound_x1, bbox[Y0_IDX]]之间的box有相交
|
||||
|
||||
if not any([b[X0_IDX] <= min_x0-1 <= b[X1_IDX] or b[X0_IDX] <= max_x1+1 <= b[X1_IDX] for b in bbox_in_bound_check]):
|
||||
# 其上,下都不能被扩展成行,暂时只检查一下上方 TODO
|
||||
top_nearest_bbox = find_all_top_bbox_direct(bbox, bboxes)
|
||||
bottom_nearest_bbox = find_all_bottom_bbox_direct(bbox, bboxes)
|
||||
if not any([
|
||||
top_nearest_bbox is not None and (find_all_left_bbox_direct(top_nearest_bbox, bboxes) is None and find_all_right_bbox_direct(top_nearest_bbox, bboxes) is None),
|
||||
bottom_nearest_bbox is not None and (find_all_left_bbox_direct(bottom_nearest_bbox, bboxes) is None and find_all_right_bbox_direct(bottom_nearest_bbox, bboxes) is None),
|
||||
top_nearest_bbox is None or bottom_nearest_bbox is None
|
||||
]):
|
||||
is_belong_to_col = True
|
||||
|
||||
# 检查是否能被下方col吸收 TODO
|
||||
|
||||
"""
|
||||
这里为什么没有is_cross_boundry_mid_line的条件呢?
|
||||
确实有些杂志左右两栏宽度不是对称的。
|
||||
"""
|
||||
if not is_belong_to_col or is_cross_boundry_mid_line:
|
||||
bbox[X0_EXT_IDX] = bound_x0
|
||||
bbox[Y0_EXT_IDX] = bbox[Y0_IDX]
|
||||
bbox[X1_EXT_IDX] = bound_x1
|
||||
bbox[Y1_EXT_IDX] = bbox[Y1_IDX]
|
||||
last_h_split_line_y1 = bbox[Y1_IDX] # 更新这条线
|
||||
else:
|
||||
continue
|
||||
"""
|
||||
此时独占一行的被成功扩展到指定的边界上,这个时候利用边界条件合并连续的bbox,成为一个group
|
||||
然后合并所有连续水平方向的bbox.
|
||||
"""
|
||||
all_bboxes.sort(key=lambda x: x[Y0_IDX])
|
||||
h_bboxes = []
|
||||
h_bbox_group = []
|
||||
|
||||
for bbox in all_bboxes:
|
||||
if bbox[X0_EXT_IDX] == bound_x0 and bbox[X1_EXT_IDX] == bound_x1:
|
||||
h_bbox_group.append(bbox)
|
||||
else:
|
||||
if len(h_bbox_group)>0:
|
||||
h_bboxes.append(h_bbox_group)
|
||||
h_bbox_group = []
|
||||
# 最后一个group
|
||||
if len(h_bbox_group)>0:
|
||||
h_bboxes.append(h_bbox_group)
|
||||
|
||||
"""
|
||||
现在h_bboxes里面是所有的group了,每个group都是一个list
|
||||
对h_bboxes里的每个group进行计算放回到sorted_layouts里
|
||||
"""
|
||||
h_layouts = []
|
||||
for gp in h_bboxes:
|
||||
gp.sort(key=lambda x: x[Y0_IDX])
|
||||
# 然后计算这个group的layout_bbox,也就是最小的x0,y0, 最大的x1,y1
|
||||
x0, y0, x1, y1 = gp[0][X0_EXT_IDX], gp[0][Y0_EXT_IDX], gp[-1][X1_EXT_IDX], gp[-1][Y1_EXT_IDX]
|
||||
h_layouts.append([x0, y0, x1, y1, LAYOUT_H]) # 水平的布局
|
||||
|
||||
"""
|
||||
接下来利用这些连续的水平bbox的layout_bbox的y0, y1,从水平上切分开其余的为几个部分
|
||||
"""
|
||||
h_split_lines = [bound_y0]
|
||||
for gp in h_bboxes: # gp是一个list[bbox_list]
|
||||
y0, y1 = gp[0][1], gp[-1][3]
|
||||
h_split_lines.append(y0)
|
||||
h_split_lines.append(y1)
|
||||
h_split_lines.append(bound_y1)
|
||||
|
||||
unsplited_bboxes = []
|
||||
for i in range(0, len(h_split_lines), 2):
|
||||
start_y0, start_y1 = h_split_lines[i:i+2]
|
||||
# 然后找出[start_y0, start_y1]之间的其他bbox,这些组成一个未分割板块
|
||||
bboxes_in_block = [bbox for bbox in all_bboxes if bbox[Y0_IDX]>=start_y0 and bbox[Y1_IDX]<=start_y1]
|
||||
unsplited_bboxes.append(bboxes_in_block)
|
||||
# 接着把未处理的加入到h_layouts里
|
||||
for bboxes_in_block in unsplited_bboxes:
|
||||
if len(bboxes_in_block) == 0:
|
||||
continue
|
||||
x0, y0, x1, y1 = bound_x0, min([bbox[Y0_IDX] for bbox in bboxes_in_block]), bound_x1, max([bbox[Y1_IDX] for bbox in bboxes_in_block])
|
||||
h_layouts.append([x0, y0, x1, y1, LAYOUT_UNPROC])
|
||||
|
||||
h_layouts.sort(key=lambda x: x[1]) # 按照y0排序, 也就是从上到下的顺序
|
||||
|
||||
"""
|
||||
转换成如下格式返回
|
||||
"""
|
||||
for layout in h_layouts:
|
||||
sorted_layout_blocks.append({
|
||||
"layout_bbox": layout[:4],
|
||||
"layout_label":layout[4],
|
||||
"sub_layout":[],
|
||||
})
|
||||
return sorted_layout_blocks
|
||||
|
||||
###############################################################################################
|
||||
#
|
||||
# 垂直方向的处理
|
||||
#
|
||||
#
|
||||
###############################################################################################
|
||||
def _vertical_align_split_v1(bboxes:list, boundry:tuple)-> list:
|
||||
"""
|
||||
计算垂直方向上的对齐, 并分割bboxes成layout。负责对一列多行的进行列维度分割。
|
||||
如果不能完全分割,剩余部分作为layout_lable为u的layout返回
|
||||
-----------------------
|
||||
| | |
|
||||
| | |
|
||||
| | |
|
||||
| | |
|
||||
-------------------------
|
||||
此函数会将:以上布局将会切分出来2列
|
||||
"""
|
||||
sorted_layout_blocks = [] # 这是要最终返回的值
|
||||
new_boundry = [boundry[0], boundry[1], boundry[2], boundry[3]]
|
||||
|
||||
v_blocks = []
|
||||
"""
|
||||
先从左到右切分
|
||||
"""
|
||||
while True:
|
||||
all_bboxes = get_bbox_in_boundry(bboxes, new_boundry)
|
||||
left_edge_bboxes = get_left_edge_bboxes(all_bboxes)
|
||||
if len(left_edge_bboxes) == 0:
|
||||
break
|
||||
right_split_line_x1 = max([bbox[X1_IDX] for bbox in left_edge_bboxes])+1
|
||||
# 然后检查这条线能不与其他bbox的左边界相交或者重合
|
||||
if any([bbox[X0_IDX] <= right_split_line_x1 <= bbox[X1_IDX] for bbox in all_bboxes]):
|
||||
# 垂直切分线与某些box发生相交,说明无法完全垂直方向切分。
|
||||
break
|
||||
else: # 说明成功分割出一列
|
||||
# 找到左侧边界最靠左的bbox作为layout的x0
|
||||
layout_x0 = min([bbox[X0_IDX] for bbox in left_edge_bboxes]) # 这里主要是为了画出来有一定间距
|
||||
v_blocks.append([layout_x0, new_boundry[1], right_split_line_x1, new_boundry[3], LAYOUT_V])
|
||||
new_boundry[0] = right_split_line_x1 # 更新边界
|
||||
|
||||
"""
|
||||
再从右到左切, 此时如果还是无法完全切分,那么剩余部分作为layout_lable为u的layout返回
|
||||
"""
|
||||
unsplited_block = []
|
||||
while True:
|
||||
all_bboxes = get_bbox_in_boundry(bboxes, new_boundry)
|
||||
right_edge_bboxes = get_right_edge_bboxes(all_bboxes)
|
||||
if len(right_edge_bboxes) == 0:
|
||||
break
|
||||
left_split_line_x0 = min([bbox[X0_IDX] for bbox in right_edge_bboxes])-1
|
||||
# 然后检查这条线能不与其他bbox的左边界相交或者重合
|
||||
if any([bbox[X0_IDX] <= left_split_line_x0 <= bbox[X1_IDX] for bbox in all_bboxes]):
|
||||
# 这里是余下的
|
||||
unsplited_block.append([new_boundry[0], new_boundry[1], new_boundry[2], new_boundry[3], LAYOUT_UNPROC])
|
||||
break
|
||||
else:
|
||||
# 找到右侧边界最靠右的bbox作为layout的x1
|
||||
layout_x1 = max([bbox[X1_IDX] for bbox in right_edge_bboxes])
|
||||
v_blocks.append([left_split_line_x0, new_boundry[1], layout_x1, new_boundry[3], LAYOUT_V])
|
||||
new_boundry[2] = left_split_line_x0 # 更新右边界
|
||||
|
||||
"""
|
||||
最后拼装成layout格式返回
|
||||
"""
|
||||
for block in v_blocks:
|
||||
sorted_layout_blocks.append({
|
||||
"layout_bbox": block[:4],
|
||||
"layout_label":block[4],
|
||||
"sub_layout":[],
|
||||
})
|
||||
for block in unsplited_block:
|
||||
sorted_layout_blocks.append({
|
||||
"layout_bbox": block[:4],
|
||||
"layout_label":block[4],
|
||||
"sub_layout":[],
|
||||
})
|
||||
|
||||
# 按照x0排序
|
||||
sorted_layout_blocks.sort(key=lambda x: x['layout_bbox'][0])
|
||||
return sorted_layout_blocks
|
||||
|
||||
def _vertical_align_split_v2(bboxes:list, boundry:tuple)-> list:
|
||||
"""
|
||||
改进的 _vertical_align_split算法,原算法会因为第二列的box由于左侧没有遮挡被认为是左侧的一部分,导致整个layout多列被识别为一列。
|
||||
利用从左上角的box开始向下看的方法,不断扩展w_x0, w_x1,直到不能继续向下扩展,或者到达边界下边界。
|
||||
"""
|
||||
sorted_layout_blocks = [] # 这是要最终返回的值
|
||||
new_boundry = [boundry[0], boundry[1], boundry[2], boundry[3]]
|
||||
bad_boxes = [] # 被割中的box
|
||||
v_blocks = []
|
||||
while True:
|
||||
all_bboxes = get_bbox_in_boundry(bboxes, new_boundry)
|
||||
if len(all_bboxes) == 0:
|
||||
break
|
||||
left_top_box = min(all_bboxes, key=lambda x: (x[X0_IDX],x[Y0_IDX]))# 这里应该加强,检查一下必须是在第一列的 TODO
|
||||
start_box = [left_top_box[X0_IDX], left_top_box[Y0_IDX], left_top_box[X1_IDX], left_top_box[Y1_IDX]]
|
||||
w_x0, w_x1 = left_top_box[X0_IDX], left_top_box[X1_IDX]
|
||||
"""
|
||||
然后沿着这个box线向下找最近的那个box, 然后扩展w_x0, w_x1
|
||||
扩展之后,宽度会增加,随后用x=w_x1来检测在边界内是否有box与相交,如果相交,那么就说明不能再扩展了。
|
||||
当不能扩展的时候就要看是否到达下边界:
|
||||
1. 达到,那么更新左边界继续分下一个列
|
||||
2. 没有达到,那么此时开始从右侧切分进入下面的循环里
|
||||
"""
|
||||
while left_top_box is not None: # 向下去找
|
||||
virtual_box = [w_x0, left_top_box[Y0_IDX], w_x1, left_top_box[Y1_IDX]]
|
||||
left_top_box = find_bottom_bbox_direct_from_left_edge(virtual_box, all_bboxes)
|
||||
if left_top_box:
|
||||
w_x0, w_x1 = min(virtual_box[X0_IDX], left_top_box[X0_IDX]), max([virtual_box[X1_IDX], left_top_box[X1_IDX]])
|
||||
# 万一这个初始的box在column中间,那么还要向上看
|
||||
start_box = [w_x0, start_box[Y0_IDX], w_x1, start_box[Y1_IDX]] # 扩展一下宽度更鲁棒
|
||||
left_top_box = find_top_bbox_direct_from_left_edge(start_box, all_bboxes)
|
||||
while left_top_box is not None: # 向上去找
|
||||
virtual_box = [w_x0, left_top_box[Y0_IDX], w_x1, left_top_box[Y1_IDX]]
|
||||
left_top_box = find_top_bbox_direct_from_left_edge(virtual_box, all_bboxes)
|
||||
if left_top_box:
|
||||
w_x0, w_x1 = min(virtual_box[X0_IDX], left_top_box[X0_IDX]), max([virtual_box[X1_IDX], left_top_box[X1_IDX]])
|
||||
|
||||
# 检查相交
|
||||
if any([bbox[X0_IDX] <= w_x1+1 <= bbox[X1_IDX] for bbox in all_bboxes]):
|
||||
for b in all_bboxes:
|
||||
if b[X0_IDX] <= w_x1+1 <= b[X1_IDX]:
|
||||
bad_boxes.append([b[X0_IDX], b[Y0_IDX], b[X1_IDX], b[Y1_IDX]])
|
||||
break
|
||||
else: # 说明成功分割出一列
|
||||
v_blocks.append([w_x0, new_boundry[1], w_x1, new_boundry[3], LAYOUT_V])
|
||||
new_boundry[0] = w_x1 # 更新边界
|
||||
|
||||
"""
|
||||
接着开始从右上角的box扫描
|
||||
"""
|
||||
w_x0 , w_x1 = 0, 0
|
||||
unsplited_block = []
|
||||
while True:
|
||||
all_bboxes = get_bbox_in_boundry(bboxes, new_boundry)
|
||||
if len(all_bboxes) == 0:
|
||||
break
|
||||
# 先找到X1最大的
|
||||
bbox_list_sorted = sorted(all_bboxes, key=lambda bbox: bbox[X1_IDX], reverse=True)
|
||||
# Then, find the boxes with the smallest Y0 value
|
||||
bigest_x1 = bbox_list_sorted[0][X1_IDX]
|
||||
boxes_with_bigest_x1 = [bbox for bbox in bbox_list_sorted if bbox[X1_IDX] == bigest_x1] # 也就是最靠右的那些
|
||||
right_top_box = min(boxes_with_bigest_x1, key=lambda bbox: bbox[Y0_IDX]) # y0最小的那个
|
||||
start_box = [right_top_box[X0_IDX], right_top_box[Y0_IDX], right_top_box[X1_IDX], right_top_box[Y1_IDX]]
|
||||
w_x0, w_x1 = right_top_box[X0_IDX], right_top_box[X1_IDX]
|
||||
|
||||
while right_top_box is not None:
|
||||
virtual_box = [w_x0, right_top_box[Y0_IDX], w_x1, right_top_box[Y1_IDX]]
|
||||
right_top_box = find_bottom_bbox_direct_from_right_edge(virtual_box, all_bboxes)
|
||||
if right_top_box:
|
||||
w_x0, w_x1 = min([w_x0, right_top_box[X0_IDX]]), max([w_x1, right_top_box[X1_IDX]])
|
||||
# 在向上扫描
|
||||
start_box = [w_x0, start_box[Y0_IDX], w_x1, start_box[Y1_IDX]] # 扩展一下宽度更鲁棒
|
||||
right_top_box = find_top_bbox_direct_from_right_edge(start_box, all_bboxes)
|
||||
while right_top_box is not None:
|
||||
virtual_box = [w_x0, right_top_box[Y0_IDX], w_x1, right_top_box[Y1_IDX]]
|
||||
right_top_box = find_top_bbox_direct_from_right_edge(virtual_box, all_bboxes)
|
||||
if right_top_box:
|
||||
w_x0, w_x1 = min([w_x0, right_top_box[X0_IDX]]), max([w_x1, right_top_box[X1_IDX]])
|
||||
|
||||
# 检查是否与其他box相交, 垂直切分线与某些box发生相交,说明无法完全垂直方向切分。
|
||||
if any([bbox[X0_IDX] <= w_x0-1 <= bbox[X1_IDX] for bbox in all_bboxes]):
|
||||
unsplited_block.append([new_boundry[0], new_boundry[1], new_boundry[2], new_boundry[3], LAYOUT_UNPROC])
|
||||
for b in all_bboxes:
|
||||
if b[X0_IDX] <= w_x0-1 <= b[X1_IDX]:
|
||||
bad_boxes.append([b[X0_IDX], b[Y0_IDX], b[X1_IDX], b[Y1_IDX]])
|
||||
break
|
||||
else: # 说明成功分割出一列
|
||||
v_blocks.append([w_x0, new_boundry[1], w_x1, new_boundry[3], LAYOUT_V])
|
||||
new_boundry[2] = w_x0
|
||||
|
||||
"""转换数据结构"""
|
||||
for block in v_blocks:
|
||||
sorted_layout_blocks.append({
|
||||
"layout_bbox": block[:4],
|
||||
"layout_label":block[4],
|
||||
"sub_layout":[],
|
||||
})
|
||||
|
||||
for block in unsplited_block:
|
||||
sorted_layout_blocks.append({
|
||||
"layout_bbox": block[:4],
|
||||
"layout_label":block[4],
|
||||
"sub_layout":[],
|
||||
"bad_boxes": bad_boxes # 记录下来,这个box是被割中的
|
||||
})
|
||||
|
||||
|
||||
# 按照x0排序
|
||||
sorted_layout_blocks.sort(key=lambda x: x['layout_bbox'][0])
|
||||
return sorted_layout_blocks
|
||||
|
||||
|
||||
|
||||
|
||||
def _try_horizontal_mult_column_split(bboxes:list, boundry:tuple)-> list:
|
||||
"""
|
||||
尝试水平切分,如果切分不动,那就当一个BAD_LAYOUT返回
|
||||
------------------
|
||||
| | |
|
||||
------------------
|
||||
| | | | <- 这里是此函数要切分的场景
|
||||
------------------
|
||||
| | |
|
||||
| | |
|
||||
"""
|
||||
pass
|
||||
|
||||
|
||||
|
||||
|
||||
def _vertical_split(bboxes:list, boundry:tuple)-> list:
|
||||
"""
|
||||
从垂直方向进行切割,分block
|
||||
这个版本里,如果垂直切分不动,那就当一个BAD_LAYOUT返回
|
||||
|
||||
--------------------------
|
||||
| | |
|
||||
| | |
|
||||
| |
|
||||
这种列是此函数要切分的 -> | |
|
||||
| |
|
||||
| | |
|
||||
| | |
|
||||
-------------------------
|
||||
"""
|
||||
sorted_layout_blocks = [] # 这是要最终返回的值
|
||||
|
||||
bound_x0, bound_y0, bound_x1, bound_y1 = boundry
|
||||
all_bboxes = get_bbox_in_boundry(bboxes, boundry)
|
||||
"""
|
||||
all_bboxes = fix_vertical_bbox_pos(all_bboxes) # 垂直方向解覆盖
|
||||
all_bboxes = fix_hor_bbox_pos(all_bboxes) # 水平解覆盖
|
||||
|
||||
这两行代码目前先不执行,因为公式检测,表格检测还不是很成熟,导致非常多的textblock参与了运算,时间消耗太大。
|
||||
这两行代码的作用是:
|
||||
如果遇到互相重叠的bbox, 那么会把面积较小的box进行压缩,从而避免重叠。对布局切分来说带来正反馈。
|
||||
"""
|
||||
|
||||
#all_bboxes = paper_bbox_sort(all_bboxes, abs(bound_x1-bound_x0), abs(bound_y1-bound_x0)) # 大致拍下序, 这个是基于直接遮挡的。
|
||||
"""
|
||||
首先在垂直方向上扩展独占一行的bbox
|
||||
|
||||
"""
|
||||
for bbox in all_bboxes:
|
||||
top_nearest_bbox = find_all_top_bbox_direct(bbox, all_bboxes) # 非扩展线
|
||||
bottom_nearest_bbox = find_all_bottom_bbox_direct(bbox, all_bboxes)
|
||||
if top_nearest_bbox is None and bottom_nearest_bbox is None and not any([b[X0_IDX]<bbox[X1_IDX]<b[X1_IDX] or b[X0_IDX]<bbox[X0_IDX]<b[X1_IDX] for b in all_bboxes]): # 独占一列, 且不和其他重叠
|
||||
bbox[X0_EXT_IDX] = bbox[X0_IDX]
|
||||
bbox[Y0_EXT_IDX] = bound_y0
|
||||
bbox[X1_EXT_IDX] = bbox[X1_IDX]
|
||||
bbox[Y1_EXT_IDX] = bound_y1
|
||||
|
||||
"""
|
||||
此时独占一列的被成功扩展到指定的边界上,这个时候利用边界条件合并连续的bbox,成为一个group
|
||||
然后合并所有连续垂直方向的bbox.
|
||||
"""
|
||||
all_bboxes.sort(key=lambda x: x[X0_IDX])
|
||||
# fix: 这里水平方向的列不要合并成一个行,因为需要保证返回给下游的最小block,总是可以无脑从上到下阅读文字。
|
||||
v_bboxes = []
|
||||
for box in all_bboxes:
|
||||
if box[Y0_EXT_IDX] == bound_y0 and box[Y1_EXT_IDX] == bound_y1:
|
||||
v_bboxes.append(box)
|
||||
|
||||
"""
|
||||
现在v_bboxes里面是所有的group了,每个group都是一个list
|
||||
对v_bboxes里的每个group进行计算放回到sorted_layouts里
|
||||
"""
|
||||
v_layouts = []
|
||||
for vbox in v_bboxes:
|
||||
#gp.sort(key=lambda x: x[X0_IDX])
|
||||
# 然后计算这个group的layout_bbox,也就是最小的x0,y0, 最大的x1,y1
|
||||
x0, y0, x1, y1 = vbox[X0_EXT_IDX], vbox[Y0_EXT_IDX], vbox[X1_EXT_IDX], vbox[Y1_EXT_IDX]
|
||||
v_layouts.append([x0, y0, x1, y1, LAYOUT_V]) # 垂直的布局
|
||||
|
||||
"""
|
||||
接下来利用这些连续的垂直bbox的layout_bbox的x0, x1,从垂直上切分开其余的为几个部分
|
||||
"""
|
||||
v_split_lines = [bound_x0]
|
||||
for gp in v_bboxes:
|
||||
x0, x1 = gp[X0_IDX], gp[X1_IDX]
|
||||
v_split_lines.append(x0)
|
||||
v_split_lines.append(x1)
|
||||
v_split_lines.append(bound_x1)
|
||||
|
||||
unsplited_bboxes = []
|
||||
for i in range(0, len(v_split_lines), 2):
|
||||
start_x0, start_x1 = v_split_lines[i:i+2]
|
||||
# 然后找出[start_x0, start_x1]之间的其他bbox,这些组成一个未分割板块
|
||||
bboxes_in_block = [bbox for bbox in all_bboxes if bbox[X0_IDX]>=start_x0 and bbox[X1_IDX]<=start_x1]
|
||||
unsplited_bboxes.append(bboxes_in_block)
|
||||
# 接着把未处理的加入到v_layouts里
|
||||
for bboxes_in_block in unsplited_bboxes:
|
||||
if len(bboxes_in_block) == 0:
|
||||
continue
|
||||
x0, y0, x1, y1 = min([bbox[X0_IDX] for bbox in bboxes_in_block]), bound_y0, max([bbox[X1_IDX] for bbox in bboxes_in_block]), bound_y1
|
||||
v_layouts.append([x0, y0, x1, y1, LAYOUT_UNPROC]) # 说明这篇区域未能够分析出可靠的版面
|
||||
|
||||
v_layouts.sort(key=lambda x: x[0]) # 按照x0排序, 也就是从左到右的顺序
|
||||
|
||||
for layout in v_layouts:
|
||||
sorted_layout_blocks.append({
|
||||
"layout_bbox": layout[:4],
|
||||
"layout_label":layout[4],
|
||||
"sub_layout":[],
|
||||
})
|
||||
|
||||
"""
|
||||
至此,垂直方向切成了2种类型,其一是独占一列的,其二是未处理的。
|
||||
下面对这些未处理的进行垂直方向切分,这个切分要切出来类似“吕”这种类型的垂直方向的布局
|
||||
"""
|
||||
for i, layout in enumerate(sorted_layout_blocks):
|
||||
if layout['layout_label'] == LAYOUT_UNPROC:
|
||||
x0, y0, x1, y1 = layout['layout_bbox']
|
||||
v_split_layouts = _vertical_align_split_v2(bboxes, [x0, y0, x1, y1])
|
||||
sorted_layout_blocks[i] = {
|
||||
"layout_bbox": [x0, y0, x1, y1],
|
||||
"layout_label": LAYOUT_H,
|
||||
"sub_layout": v_split_layouts
|
||||
}
|
||||
layout['layout_label'] = LAYOUT_H # 被垂线切分成了水平布局
|
||||
|
||||
return sorted_layout_blocks
|
||||
|
||||
|
||||
def split_layout(bboxes:list, boundry:tuple, page_num:int)-> list:
|
||||
"""
|
||||
把bboxes切割成layout
|
||||
return:
|
||||
[
|
||||
{
|
||||
"layout_bbox": [x0, y0, x1, y1],
|
||||
"layout_label":"u|v|h|b", 未处理|垂直|水平|BAD_LAYOUT
|
||||
"sub_layout": [] #每个元素都是[x0, y0, x1, y1, block_content, idx_x, idx_y, content_type, ext_x0, ext_y0, ext_x1, ext_y1], 并且顺序就是阅读顺序
|
||||
}
|
||||
]
|
||||
example:
|
||||
[
|
||||
{
|
||||
"layout_bbox": [0, 0, 100, 100],
|
||||
"layout_label":"u|v|h|b",
|
||||
"sub_layout":[
|
||||
|
||||
]
|
||||
},
|
||||
{
|
||||
"layout_bbox": [0, 0, 100, 100],
|
||||
"layout_label":"u|v|h|b",
|
||||
"sub_layout":[
|
||||
{
|
||||
"layout_bbox": [0, 0, 100, 100],
|
||||
"layout_label":"u|v|h|b",
|
||||
"content_bboxes":[
|
||||
[],
|
||||
[],
|
||||
[]
|
||||
]
|
||||
},
|
||||
{
|
||||
"layout_bbox": [0, 0, 100, 100],
|
||||
"layout_label":"u|v|h|b",
|
||||
"sub_layout":[
|
||||
|
||||
]
|
||||
}
|
||||
}
|
||||
]
|
||||
"""
|
||||
sorted_layouts = [] # 最终返回的结果
|
||||
|
||||
boundry_x0, boundry_y0, boundry_x1, boundry_y1 = boundry
|
||||
if len(bboxes) <=1:
|
||||
return [
|
||||
{
|
||||
"layout_bbox": [boundry_x0, boundry_y0, boundry_x1, boundry_y1],
|
||||
"layout_label": LAYOUT_V,
|
||||
"sub_layout":[]
|
||||
}
|
||||
]
|
||||
|
||||
"""
|
||||
接下来按照先水平后垂直的顺序进行切分
|
||||
"""
|
||||
bboxes = paper_bbox_sort(bboxes, boundry_x1-boundry_x0, boundry_y1-boundry_y0)
|
||||
sorted_layouts = _horizontal_split(bboxes, boundry) # 通过水平分割出来的layout
|
||||
for i, layout in enumerate(sorted_layouts):
|
||||
x0, y0, x1, y1 = layout['layout_bbox']
|
||||
layout_type = layout['layout_label']
|
||||
if layout_type == LAYOUT_UNPROC: # 说明是非独占单行的,这些需要垂直切分
|
||||
v_split_layouts = _vertical_split(bboxes, [x0, y0, x1, y1])
|
||||
|
||||
"""
|
||||
最后这里有个逻辑问题:如果这个函数只分离出来了一个column layout,那么这个layout分割肯定超出了算法能力范围。因为我们假定的是传进来的
|
||||
box已经把行全部剥离了,所以这里必须十多个列才可以。如果只剥离出来一个layout,并且是多个box,那么就说明这个layout是无法分割的,标记为LAYOUT_UNPROC
|
||||
"""
|
||||
layout_label = LAYOUT_V
|
||||
if len(v_split_layouts) == 1:
|
||||
if len(v_split_layouts[0]['sub_layout']) == 0:
|
||||
layout_label = LAYOUT_UNPROC
|
||||
#logger.warning(f"WARNING: pageno={page_num}, 无法分割的layout: ", v_split_layouts)
|
||||
|
||||
"""
|
||||
组合起来最终的layout
|
||||
"""
|
||||
sorted_layouts[i] = {
|
||||
"layout_bbox": [x0, y0, x1, y1],
|
||||
"layout_label": layout_label,
|
||||
"sub_layout": v_split_layouts
|
||||
}
|
||||
layout['layout_label'] = LAYOUT_H
|
||||
|
||||
"""
|
||||
水平和垂直方向都切分完毕了。此时还有一些未处理的,这些未处理的可能是因为水平和垂直方向都无法切分。
|
||||
这些最后调用_try_horizontal_mult_block_split做一次水平多个block的联合切分,如果也不能切分最终就当做BAD_LAYOUT返回
|
||||
"""
|
||||
# TODO
|
||||
|
||||
return sorted_layouts
|
||||
|
||||
|
||||
def get_bboxes_layout(all_boxes:list, boundry:tuple, page_id:int):
|
||||
"""
|
||||
对利用layout排序之后的box,进行排序
|
||||
return:
|
||||
[
|
||||
{
|
||||
"layout_bbox": [x0, y0, x1, y1],
|
||||
"layout_label":"u|v|h|b", 未处理|垂直|水平|BAD_LAYOUT
|
||||
},
|
||||
]
|
||||
"""
|
||||
def _preorder_traversal(layout):
|
||||
"""
|
||||
对sorted_layouts的叶子节点,也就是len(sub_layout)==0的节点进行排序。排序按照前序遍历的顺序,也就是从上到下,从左到右的顺序
|
||||
"""
|
||||
sorted_layout_blocks = []
|
||||
for layout in layout:
|
||||
sub_layout = layout['sub_layout']
|
||||
if len(sub_layout) == 0:
|
||||
sorted_layout_blocks.append(layout)
|
||||
else:
|
||||
s = _preorder_traversal(sub_layout)
|
||||
sorted_layout_blocks.extend(s)
|
||||
return sorted_layout_blocks
|
||||
# -------------------------------------------------------------------------------------------------------------------------
|
||||
sorted_layouts = split_layout(all_boxes, boundry, page_id)# 先切分成layout,得到一个Tree
|
||||
total_sorted_layout_blocks = _preorder_traversal(sorted_layouts)
|
||||
return total_sorted_layout_blocks, sorted_layouts
|
||||
|
||||
|
||||
def get_columns_cnt_of_layout(layout_tree):
|
||||
"""
|
||||
获取一个layout的宽度
|
||||
"""
|
||||
max_width_list = [0] # 初始化一个元素,防止max,min函数报错
|
||||
|
||||
for items in layout_tree: # 针对每一层(横切)计算列数,横着的算一列
|
||||
layout_type = items['layout_label']
|
||||
sub_layouts = items['sub_layout']
|
||||
if len(sub_layouts)==0:
|
||||
max_width_list.append(1)
|
||||
else:
|
||||
if layout_type == LAYOUT_H:
|
||||
max_width_list.append(1)
|
||||
else:
|
||||
width = 0
|
||||
for l in sub_layouts:
|
||||
if len(l['sub_layout']) == 0:
|
||||
width += 1
|
||||
else:
|
||||
for lay in l['sub_layout']:
|
||||
width += get_columns_cnt_of_layout([lay])
|
||||
max_width_list.append(width)
|
||||
|
||||
return max(max_width_list)
|
||||
|
||||
|
||||
|
||||
def sort_with_layout(bboxes:list, page_width, page_height) -> (list,list):
|
||||
"""
|
||||
输入是一个bbox的list.
|
||||
获取到输入之后,先进行layout切分,然后对这些bbox进行排序。返回排序后的bboxes
|
||||
"""
|
||||
|
||||
new_bboxes = []
|
||||
for box in bboxes:
|
||||
# new_bboxes.append([box[0], box[1], box[2], box[3], None, None, None, 'text', None, None, None, None])
|
||||
new_bboxes.append([box[0], box[1], box[2], box[3], None, None, None, 'text', None, None, None, None, box[4]])
|
||||
|
||||
layout_bboxes, _ = get_bboxes_layout(new_bboxes, [0, 0, page_width, page_height], 0)
|
||||
if any([lay['layout_label']==LAYOUT_UNPROC for lay in layout_bboxes]):
|
||||
logger.warning(f"drop this pdf, reason: 复杂版面")
|
||||
return None,None
|
||||
|
||||
sorted_bboxes = []
|
||||
# 利用layout bbox每次框定一些box,然后排序
|
||||
for layout in layout_bboxes:
|
||||
lbox = layout['layout_bbox']
|
||||
bbox_in_layout = get_bbox_in_boundry(new_bboxes, lbox)
|
||||
sorted_bbox = paper_bbox_sort(bbox_in_layout, lbox[2]-lbox[0], lbox[3]-lbox[1])
|
||||
sorted_bboxes.extend(sorted_bbox)
|
||||
|
||||
return sorted_bboxes, layout_bboxes
|
||||
|
||||
|
||||
def sort_text_block(text_block, layout_bboxes):
|
||||
"""
|
||||
对一页的text_block进行排序
|
||||
"""
|
||||
sorted_text_bbox = []
|
||||
all_text_bbox = []
|
||||
# 做一个box=>text的映射
|
||||
box_to_text = {}
|
||||
for blk in text_block:
|
||||
box = blk['bbox']
|
||||
box_to_text[(box[0], box[1], box[2], box[3])] = blk
|
||||
all_text_bbox.append(box)
|
||||
|
||||
# text_blocks_to_sort = []
|
||||
# for box in box_to_text.keys():
|
||||
# text_blocks_to_sort.append([box[0], box[1], box[2], box[3], None, None, None, 'text', None, None, None, None])
|
||||
|
||||
# 按照layout_bboxes的顺序,对text_block进行排序
|
||||
for layout in layout_bboxes:
|
||||
layout_box = layout['layout_bbox']
|
||||
text_bbox_in_layout = get_bbox_in_boundry(all_text_bbox, [layout_box[0]-1, layout_box[1]-1, layout_box[2]+1, layout_box[3]+1])
|
||||
#sorted_bbox = paper_bbox_sort(text_bbox_in_layout, layout_box[2]-layout_box[0], layout_box[3]-layout_box[1])
|
||||
text_bbox_in_layout.sort(key = lambda x: x[1]) # 一个layout内部的box,按照y0自上而下排序
|
||||
#sorted_bbox = [[b] for b in text_blocks_to_sort]
|
||||
for sb in text_bbox_in_layout:
|
||||
sorted_text_bbox.append(box_to_text[(sb[0], sb[1], sb[2], sb[3])])
|
||||
|
||||
return sorted_text_bbox
|
||||
@@ -0,0 +1,100 @@
|
||||
"""
|
||||
找到能分割布局的水平的横线、色块
|
||||
"""
|
||||
|
||||
import os, fitz
|
||||
from libs.boxbase import _is_in_or_part_overlap
|
||||
|
||||
|
||||
def __rect_filter_by_width(rect, page_w, page_h):
|
||||
mid_x = page_w/2
|
||||
if rect[0]< mid_x < rect[2]:
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def __rect_filter_by_pos(rect, image_bboxes, table_bboxes):
|
||||
"""
|
||||
不能出现在table和image的位置
|
||||
"""
|
||||
for box in image_bboxes:
|
||||
if _is_in_or_part_overlap(rect, box):
|
||||
return False
|
||||
|
||||
for box in table_bboxes:
|
||||
if _is_in_or_part_overlap(rect, box):
|
||||
return False
|
||||
|
||||
return True
|
||||
|
||||
|
||||
def __debug_show_page(page, bboxes1: list,bboxes2: list,bboxes3: list,):
|
||||
save_path = "./tmp/debug.pdf"
|
||||
if os.path.exists(save_path):
|
||||
# 删除已经存在的文件
|
||||
os.remove(save_path)
|
||||
# 创建一个新的空白 PDF 文件
|
||||
doc = fitz.open('')
|
||||
|
||||
width = page.rect.width
|
||||
height = page.rect.height
|
||||
new_page = doc.new_page(width=width, height=height)
|
||||
|
||||
shape = new_page.new_shape()
|
||||
for bbox in bboxes1:
|
||||
# 原始box画上去
|
||||
rect = fitz.Rect(*bbox[0:4])
|
||||
shape = new_page.new_shape()
|
||||
shape.draw_rect(rect)
|
||||
shape.finish(color=fitz.pdfcolor['red'], fill=fitz.pdfcolor['blue'], fill_opacity=0.2)
|
||||
shape.finish()
|
||||
shape.commit()
|
||||
|
||||
for bbox in bboxes2:
|
||||
# 原始box画上去
|
||||
rect = fitz.Rect(*bbox[0:4])
|
||||
shape = new_page.new_shape()
|
||||
shape.draw_rect(rect)
|
||||
shape.finish(color=None, fill=fitz.pdfcolor['yellow'], fill_opacity=0.2)
|
||||
shape.finish()
|
||||
shape.commit()
|
||||
|
||||
for bbox in bboxes3:
|
||||
# 原始box画上去
|
||||
rect = fitz.Rect(*bbox[0:4])
|
||||
shape = new_page.new_shape()
|
||||
shape.draw_rect(rect)
|
||||
shape.finish(color=fitz.pdfcolor['red'], fill=None)
|
||||
shape.finish()
|
||||
shape.commit()
|
||||
|
||||
parent_dir = os.path.dirname(save_path)
|
||||
if not os.path.exists(parent_dir):
|
||||
os.makedirs(parent_dir)
|
||||
|
||||
doc.save(save_path)
|
||||
doc.close()
|
||||
|
||||
def get_spilter_of_page(page, image_bboxes, table_bboxes):
|
||||
"""
|
||||
获取到色块和横线
|
||||
"""
|
||||
cdrawings = page.get_cdrawings()
|
||||
|
||||
spilter_bbox = []
|
||||
for block in cdrawings:
|
||||
if 'fill' in block:
|
||||
fill = block['fill']
|
||||
if 'fill' in block and block['fill'] and block['fill']!=(1.0,1.0,1.0):
|
||||
rect = block['rect']
|
||||
if __rect_filter_by_width(rect, page.rect.width, page.rect.height) and __rect_filter_by_pos(rect, image_bboxes, table_bboxes):
|
||||
spilter_bbox.append(list(rect))
|
||||
|
||||
"""过滤、修正一下这些box。因为有时候会有一些矩形,高度为0或者为负数,造成layout计算无限循环。如果是负高度或者0高度,统一修正为高度为1"""
|
||||
for box in spilter_bbox:
|
||||
if box[3]-box[1] <= 0:
|
||||
box[3] = box[1] + 1
|
||||
|
||||
#__debug_show_page(page, spilter_bbox, [], [])
|
||||
|
||||
return spilter_bbox
|
||||
@@ -0,0 +1,337 @@
|
||||
"""
|
||||
This is an advanced PyMuPDF utility for detecting multi-column pages.
|
||||
It can be used in a shell script, or its main function can be imported and
|
||||
invoked as descript below.
|
||||
|
||||
Features
|
||||
---------
|
||||
- Identify text belonging to (a variable number of) columns on the page.
|
||||
- Text with different background color is handled separately, allowing for
|
||||
easier treatment of side remarks, comment boxes, etc.
|
||||
- Uses text block detection capability to identify text blocks and
|
||||
uses the block bboxes as primary structuring principle.
|
||||
- Supports ignoring footers via a footer margin parameter.
|
||||
- Returns re-created text boundary boxes (integer coordinates), sorted ascending
|
||||
by the top, then by the left coordinates.
|
||||
|
||||
Restrictions
|
||||
-------------
|
||||
- Only supporting horizontal, left-to-right text
|
||||
- Returns a list of text boundary boxes - not the text itself. The caller is
|
||||
expected to extract text from within the returned boxes.
|
||||
- Text written above images is ignored altogether (option).
|
||||
- This utility works as expected in most cases. The following situation cannot
|
||||
be handled correctly:
|
||||
* overlapping (non-disjoint) text blocks
|
||||
* image captions are not recognized and are handled like normal text
|
||||
|
||||
Usage
|
||||
------
|
||||
- As a CLI shell command use
|
||||
|
||||
python multi_column.py input.pdf footer_margin
|
||||
|
||||
Where footer margin is the height of the bottom stripe to ignore on each page.
|
||||
This code is intended to be modified according to your need.
|
||||
|
||||
- Use in a Python script as follows:
|
||||
|
||||
----------------------------------------------------------------------------------
|
||||
from multi_column import column_boxes
|
||||
|
||||
# for each page execute
|
||||
bboxes = column_boxes(page, footer_margin=50, no_image_text=True)
|
||||
|
||||
# bboxes is a list of fitz.IRect objects, that are sort ascending by their y0,
|
||||
# then x0 coordinates. Their text content can be extracted by all PyMuPDF
|
||||
# get_text() variants, like for instance the following:
|
||||
for rect in bboxes:
|
||||
print(page.get_text(clip=rect, sort=True))
|
||||
----------------------------------------------------------------------------------
|
||||
"""
|
||||
import os
|
||||
import sys
|
||||
from libs.commons import fitz
|
||||
|
||||
|
||||
def column_boxes(page, footer_margin=50, header_margin=50, no_image_text=True):
|
||||
"""Determine bboxes which wrap a column."""
|
||||
paths = page.get_drawings()
|
||||
bboxes = []
|
||||
|
||||
# path rectangles
|
||||
path_rects = []
|
||||
|
||||
# image bboxes
|
||||
img_bboxes = []
|
||||
|
||||
# bboxes of non-horizontal text
|
||||
# avoid when expanding horizontal text boxes
|
||||
vert_bboxes = []
|
||||
|
||||
# compute relevant page area
|
||||
clip = +page.rect
|
||||
clip.y1 -= footer_margin # Remove footer area
|
||||
clip.y0 += header_margin # Remove header area
|
||||
|
||||
def can_extend(temp, bb, bboxlist):
|
||||
"""Determines whether rectangle 'temp' can be extended by 'bb'
|
||||
without intersecting any of the rectangles contained in 'bboxlist'.
|
||||
|
||||
Items of bboxlist may be None if they have been removed.
|
||||
|
||||
Returns:
|
||||
True if 'temp' has no intersections with items of 'bboxlist'.
|
||||
"""
|
||||
for b in bboxlist:
|
||||
if not intersects_bboxes(temp, vert_bboxes) and (
|
||||
b == None or b == bb or (temp & b).is_empty
|
||||
):
|
||||
continue
|
||||
return False
|
||||
|
||||
return True
|
||||
|
||||
def in_bbox(bb, bboxes):
|
||||
"""Return 1-based number if a bbox contains bb, else return 0."""
|
||||
for i, bbox in enumerate(bboxes):
|
||||
if bb in bbox:
|
||||
return i + 1
|
||||
return 0
|
||||
|
||||
def intersects_bboxes(bb, bboxes):
|
||||
"""Return True if a bbox intersects bb, else return False."""
|
||||
for bbox in bboxes:
|
||||
if not (bb & bbox).is_empty:
|
||||
return True
|
||||
return False
|
||||
|
||||
def extend_right(bboxes, width, path_bboxes, vert_bboxes, img_bboxes):
|
||||
"""Extend a bbox to the right page border.
|
||||
|
||||
Whenever there is no text to the right of a bbox, enlarge it up
|
||||
to the right page border.
|
||||
|
||||
Args:
|
||||
bboxes: (list[IRect]) bboxes to check
|
||||
width: (int) page width
|
||||
path_bboxes: (list[IRect]) bboxes with a background color
|
||||
vert_bboxes: (list[IRect]) bboxes with vertical text
|
||||
img_bboxes: (list[IRect]) bboxes of images
|
||||
Returns:
|
||||
Potentially modified bboxes.
|
||||
"""
|
||||
for i, bb in enumerate(bboxes):
|
||||
# do not extend text with background color
|
||||
if in_bbox(bb, path_bboxes):
|
||||
continue
|
||||
|
||||
# do not extend text in images
|
||||
if in_bbox(bb, img_bboxes):
|
||||
continue
|
||||
|
||||
# temp extends bb to the right page border
|
||||
temp = +bb
|
||||
temp.x1 = width
|
||||
|
||||
# do not cut through colored background or images
|
||||
if intersects_bboxes(temp, path_bboxes + vert_bboxes + img_bboxes):
|
||||
continue
|
||||
|
||||
# also, do not intersect other text bboxes
|
||||
check = can_extend(temp, bb, bboxes)
|
||||
if check:
|
||||
bboxes[i] = temp # replace with enlarged bbox
|
||||
|
||||
return [b for b in bboxes if b != None]
|
||||
|
||||
def clean_nblocks(nblocks):
|
||||
"""Do some elementary cleaning."""
|
||||
|
||||
# 1. remove any duplicate blocks.
|
||||
blen = len(nblocks)
|
||||
if blen < 2:
|
||||
return nblocks
|
||||
start = blen - 1
|
||||
for i in range(start, -1, -1):
|
||||
bb1 = nblocks[i]
|
||||
bb0 = nblocks[i - 1]
|
||||
if bb0 == bb1:
|
||||
del nblocks[i]
|
||||
|
||||
# 2. repair sequence in special cases:
|
||||
# consecutive bboxes with almost same bottom value are sorted ascending
|
||||
# by x-coordinate.
|
||||
y1 = nblocks[0].y1 # first bottom coordinate
|
||||
i0 = 0 # its index
|
||||
i1 = -1 # index of last bbox with same bottom
|
||||
|
||||
# Iterate over bboxes, identifying segments with approx. same bottom value.
|
||||
# Replace every segment by its sorted version.
|
||||
for i in range(1, len(nblocks)):
|
||||
b1 = nblocks[i]
|
||||
if abs(b1.y1 - y1) > 10: # different bottom
|
||||
if i1 > i0: # segment length > 1? Sort it!
|
||||
nblocks[i0 : i1 + 1] = sorted(
|
||||
nblocks[i0 : i1 + 1], key=lambda b: b.x0
|
||||
)
|
||||
y1 = b1.y1 # store new bottom value
|
||||
i0 = i # store its start index
|
||||
i1 = i # store current index
|
||||
if i1 > i0: # segment waiting to be sorted
|
||||
nblocks[i0 : i1 + 1] = sorted(nblocks[i0 : i1 + 1], key=lambda b: b.x0)
|
||||
return nblocks
|
||||
|
||||
# extract vector graphics
|
||||
for p in paths:
|
||||
path_rects.append(p["rect"].irect)
|
||||
path_bboxes = path_rects
|
||||
|
||||
# sort path bboxes by ascending top, then left coordinates
|
||||
path_bboxes.sort(key=lambda b: (b.y0, b.x0))
|
||||
|
||||
# bboxes of images on page, no need to sort them
|
||||
for item in page.get_images():
|
||||
img_bboxes.extend(page.get_image_rects(item[0]))
|
||||
|
||||
# blocks of text on page
|
||||
blocks = page.get_text(
|
||||
"dict",
|
||||
flags=fitz.TEXTFLAGS_TEXT,
|
||||
clip=clip,
|
||||
)["blocks"]
|
||||
|
||||
# Make block rectangles, ignoring non-horizontal text
|
||||
for b in blocks:
|
||||
bbox = fitz.IRect(b["bbox"]) # bbox of the block
|
||||
|
||||
# ignore text written upon images
|
||||
if no_image_text and in_bbox(bbox, img_bboxes):
|
||||
continue
|
||||
|
||||
# confirm first line to be horizontal
|
||||
line0 = b["lines"][0] # get first line
|
||||
if line0["dir"] != (1, 0): # only accept horizontal text
|
||||
vert_bboxes.append(bbox)
|
||||
continue
|
||||
|
||||
srect = fitz.EMPTY_IRECT()
|
||||
for line in b["lines"]:
|
||||
lbbox = fitz.IRect(line["bbox"])
|
||||
text = "".join([s["text"].strip() for s in line["spans"]])
|
||||
if len(text) > 1:
|
||||
srect |= lbbox
|
||||
bbox = +srect
|
||||
|
||||
if not bbox.is_empty:
|
||||
bboxes.append(bbox)
|
||||
|
||||
# Sort text bboxes by ascending background, top, then left coordinates
|
||||
bboxes.sort(key=lambda k: (in_bbox(k, path_bboxes), k.y0, k.x0))
|
||||
|
||||
# Extend bboxes to the right where possible
|
||||
bboxes = extend_right(
|
||||
bboxes, int(page.rect.width), path_bboxes, vert_bboxes, img_bboxes
|
||||
)
|
||||
|
||||
# immediately return of no text found
|
||||
if bboxes == []:
|
||||
return []
|
||||
|
||||
# --------------------------------------------------------------------
|
||||
# Join bboxes to establish some column structure
|
||||
# --------------------------------------------------------------------
|
||||
# the final block bboxes on page
|
||||
nblocks = [bboxes[0]] # pre-fill with first bbox
|
||||
bboxes = bboxes[1:] # remaining old bboxes
|
||||
|
||||
for i, bb in enumerate(bboxes): # iterate old bboxes
|
||||
check = False # indicates unwanted joins
|
||||
|
||||
# check if bb can extend one of the new blocks
|
||||
for j in range(len(nblocks)):
|
||||
nbb = nblocks[j] # a new block
|
||||
|
||||
# never join across columns
|
||||
if bb == None or nbb.x1 < bb.x0 or bb.x1 < nbb.x0:
|
||||
continue
|
||||
|
||||
# never join across different background colors
|
||||
if in_bbox(nbb, path_bboxes) != in_bbox(bb, path_bboxes):
|
||||
continue
|
||||
|
||||
temp = bb | nbb # temporary extension of new block
|
||||
check = can_extend(temp, nbb, nblocks)
|
||||
if check == True:
|
||||
break
|
||||
|
||||
if not check: # bb cannot be used to extend any of the new bboxes
|
||||
nblocks.append(bb) # so add it to the list
|
||||
j = len(nblocks) - 1 # index of it
|
||||
temp = nblocks[j] # new bbox added
|
||||
|
||||
# check if some remaining bbox is contained in temp
|
||||
check = can_extend(temp, bb, bboxes)
|
||||
if check == False:
|
||||
nblocks.append(bb)
|
||||
else:
|
||||
nblocks[j] = temp
|
||||
bboxes[i] = None
|
||||
|
||||
# do some elementary cleaning
|
||||
nblocks = clean_nblocks(nblocks)
|
||||
|
||||
# return identified text bboxes
|
||||
return nblocks
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
"""Only for debugging purposes, currently.
|
||||
|
||||
Draw red borders around the returned text bboxes and insert
|
||||
the bbox number.
|
||||
Then save the file under the name "input-blocks.pdf".
|
||||
"""
|
||||
|
||||
# get the file name
|
||||
filename = sys.argv[1]
|
||||
|
||||
# check if footer margin is given
|
||||
if len(sys.argv) > 2:
|
||||
footer_margin = int(sys.argv[2])
|
||||
else: # use default vaue
|
||||
footer_margin = 50
|
||||
|
||||
# check if header margin is given
|
||||
if len(sys.argv) > 3:
|
||||
header_margin = int(sys.argv[3])
|
||||
else: # use default vaue
|
||||
header_margin = 50
|
||||
|
||||
# open document
|
||||
doc = fitz.open(filename)
|
||||
|
||||
# iterate over the pages
|
||||
for page in doc:
|
||||
# remove any geometry issues
|
||||
page.wrap_contents()
|
||||
|
||||
# get the text bboxes
|
||||
bboxes = column_boxes(page, footer_margin=footer_margin, header_margin=header_margin)
|
||||
|
||||
# prepare a canvas to draw rectangles and text
|
||||
shape = page.new_shape()
|
||||
|
||||
# iterate over the bboxes
|
||||
for i, rect in enumerate(bboxes):
|
||||
shape.draw_rect(rect) # draw a border
|
||||
|
||||
# write sequence number
|
||||
shape.insert_text(rect.tl + (5, 15), str(i), color=fitz.pdfcolor["red"])
|
||||
|
||||
# finish drawing / text with color red
|
||||
shape.finish(color=fitz.pdfcolor["red"])
|
||||
shape.commit() # store to the page
|
||||
|
||||
# save document with text bboxes
|
||||
doc.ez_save(filename.replace(".pdf", "-blocks.pdf"))
|
||||
+253
@@ -0,0 +1,253 @@
|
||||
|
||||
|
||||
from loguru import logger
|
||||
|
||||
|
||||
def _is_in_or_part_overlap(box1, box2) -> bool:
|
||||
"""
|
||||
两个bbox是否有部分重叠或者包含
|
||||
"""
|
||||
if box1 is None or box2 is None:
|
||||
return False
|
||||
|
||||
x0_1, y0_1, x1_1, y1_1 = box1
|
||||
x0_2, y0_2, x1_2, y1_2 = box2
|
||||
|
||||
return not (x1_1 < x0_2 or # box1在box2的左边
|
||||
x0_1 > x1_2 or # box1在box2的右边
|
||||
y1_1 < y0_2 or # box1在box2的上边
|
||||
y0_1 > y1_2) # box1在box2的下边
|
||||
|
||||
def _is_in(box1, box2) -> bool:
|
||||
"""
|
||||
box1是否完全在box2里面
|
||||
"""
|
||||
x0_1, y0_1, x1_1, y1_1 = box1
|
||||
x0_2, y0_2, x1_2, y1_2 = box2
|
||||
|
||||
return (x0_1 >= x0_2 and # box1的左边界不在box2的左边外
|
||||
y0_1 >= y0_2 and # box1的上边界不在box2的上边外
|
||||
x1_1 <= x1_2 and # box1的右边界不在box2的右边外
|
||||
y1_1 <= y1_2) # box1的下边界不在box2的下边外
|
||||
|
||||
def _is_part_overlap(box1, box2) -> bool:
|
||||
"""
|
||||
两个bbox是否有部分重叠,但不完全包含
|
||||
"""
|
||||
if box1 is None or box2 is None:
|
||||
return False
|
||||
|
||||
return _is_in_or_part_overlap(box1, box2) and not _is_in(box1, box2)
|
||||
|
||||
def _left_intersect(left_box, right_box):
|
||||
"检查两个box的左边界是否有交集,也就是left_box的右边界是否在right_box的左边界内"
|
||||
if left_box is None or right_box is None:
|
||||
return False
|
||||
|
||||
x0_1, y0_1, x1_1, y1_1 = left_box
|
||||
x0_2, y0_2, x1_2, y1_2 = right_box
|
||||
|
||||
return x1_1>x0_2 and x0_1<x0_2 and (y0_1<=y0_2<=y1_1 or y0_1<=y1_2<=y1_1)
|
||||
|
||||
def _right_intersect(left_box, right_box):
|
||||
"""
|
||||
检查box是否在右侧边界有交集,也就是left_box的左边界是否在right_box的右边界内
|
||||
"""
|
||||
if left_box is None or right_box is None:
|
||||
return False
|
||||
|
||||
x0_1, y0_1, x1_1, y1_1 = left_box
|
||||
x0_2, y0_2, x1_2, y1_2 = right_box
|
||||
|
||||
return x0_1<x1_2 and x1_1>x1_2 and (y0_1<=y0_2<=y1_1 or y0_1<=y1_2<=y1_1)
|
||||
|
||||
|
||||
def _is_vertical_full_overlap(box1, box2, x_torlence=2):
|
||||
"""
|
||||
x方向上:要么box1包含box2, 要么box2包含box1。不能部分包含
|
||||
y方向上:box1和box2有重叠
|
||||
"""
|
||||
# 解析box的坐标
|
||||
x11, y11, x12, y12 = box1 # 左上角和右下角的坐标 (x1, y1, x2, y2)
|
||||
x21, y21, x22, y22 = box2
|
||||
|
||||
# 在x轴方向上,box1是否包含box2 或 box2包含box1
|
||||
contains_in_x = (x11-x_torlence <= x21 and x12+x_torlence >= x22) or (x21-x_torlence <= x11 and x22+x_torlence >= x12)
|
||||
|
||||
# 在y轴方向上,box1和box2是否有重叠
|
||||
overlap_in_y = not (y12 < y21 or y11 > y22)
|
||||
|
||||
return contains_in_x and overlap_in_y
|
||||
|
||||
|
||||
def _is_bottom_full_overlap(box1, box2, y_tolerance=2):
|
||||
"""
|
||||
检查box1下方和box2的上方有轻微的重叠,轻微程度收到y_tolerance的限制
|
||||
这个函数和_is_vertical-full_overlap的区别是,这个函数允许box1和box2在x方向上有轻微的重叠,允许一定的模糊度
|
||||
"""
|
||||
if box1 is None or box2 is None:
|
||||
return False
|
||||
|
||||
x0_1, y0_1, x1_1, y1_1 = box1
|
||||
x0_2, y0_2, x1_2, y1_2 = box2
|
||||
tolerance_margin = 2
|
||||
is_xdir_full_overlap = ((x0_1-tolerance_margin<=x0_2<=x1_1+tolerance_margin and x0_1-tolerance_margin<=x1_2<=x1_1+tolerance_margin) or (x0_2-tolerance_margin<=x0_1<=x1_2+tolerance_margin and x0_2-tolerance_margin<=x1_1<=x1_2+tolerance_margin))
|
||||
|
||||
return y0_2<y1_1 and 0<(y1_1-y0_2)<y_tolerance and is_xdir_full_overlap
|
||||
|
||||
def _is_left_overlap(box1, box2,):
|
||||
"""
|
||||
检查box1的左侧是否和box2有重叠
|
||||
在Y方向上可以是部分重叠或者是完全重叠。不分box1和box2的上下关系,也就是无论box1在box2下方还是box2在box1下方,都可以检测到重叠。
|
||||
X方向上
|
||||
"""
|
||||
def __overlap_y(Ay1, Ay2, By1, By2):
|
||||
return max(0, min(Ay2, By2) - max(Ay1, By1))
|
||||
|
||||
if box1 is None or box2 is None:
|
||||
return False
|
||||
|
||||
x0_1, y0_1, x1_1, y1_1 = box1
|
||||
x0_2, y0_2, x1_2, y1_2 = box2
|
||||
|
||||
y_overlap_len = __overlap_y(y0_1, y1_1, y0_2, y1_2)
|
||||
ratio_1 = 1.0 * y_overlap_len / (y1_1 - y0_1) if y1_1-y0_1!=0 else 0
|
||||
ratio_2 = 1.0 * y_overlap_len / (y1_2 - y0_2) if y1_2-y0_2!=0 else 0
|
||||
vertical_overlap_cond = ratio_1 >= 0.5 or ratio_2 >= 0.5
|
||||
|
||||
#vertical_overlap_cond = y0_1<=y0_2<=y1_1 or y0_1<=y1_2<=y1_1 or y0_2<=y0_1<=y1_2 or y0_2<=y1_1<=y1_2
|
||||
return x0_1<=x0_2<=x1_1 and vertical_overlap_cond
|
||||
|
||||
|
||||
def calculate_iou(bbox1, bbox2):
|
||||
# Determine the coordinates of the intersection rectangle
|
||||
x_left = max(bbox1[0], bbox2[0])
|
||||
y_top = max(bbox1[1], bbox2[1])
|
||||
x_right = min(bbox1[2], bbox2[2])
|
||||
y_bottom = min(bbox1[3], bbox2[3])
|
||||
|
||||
if x_right < x_left or y_bottom < y_top:
|
||||
return 0.0
|
||||
|
||||
# The area of overlap area
|
||||
intersection_area = (x_right - x_left) * (y_bottom - y_top)
|
||||
|
||||
# The area of both rectangles
|
||||
bbox1_area = (bbox1[2] - bbox1[0]) * (bbox1[3] - bbox1[1])
|
||||
bbox2_area = (bbox2[2] - bbox2[0]) * (bbox2[3] - bbox2[1])
|
||||
|
||||
# Compute the intersection over union by taking the intersection area
|
||||
# and dividing it by the sum of both areas minus the intersection area
|
||||
iou = intersection_area / float(bbox1_area + bbox2_area - intersection_area)
|
||||
return iou
|
||||
|
||||
|
||||
def calculate_overlap_area_2_minbox_area_ratio(bbox1, bbox2):
|
||||
"""
|
||||
计算box1和box2的重叠面积占最小面积的box的比例
|
||||
"""
|
||||
# Determine the coordinates of the intersection rectangle
|
||||
x_left = max(bbox1[0], bbox2[0])
|
||||
y_top = max(bbox1[1], bbox2[1])
|
||||
x_right = min(bbox1[2], bbox2[2])
|
||||
y_bottom = min(bbox1[3], bbox2[3])
|
||||
|
||||
if x_right < x_left or y_bottom < y_top:
|
||||
return 0.0
|
||||
|
||||
# The area of overlap area
|
||||
intersection_area = (x_right - x_left) * (y_bottom - y_top)
|
||||
min_box_area = min([(bbox1[2]-bbox1[0])*(bbox1[3]-bbox1[1]), (bbox2[3]-bbox2[1])*(bbox2[2]-bbox2[0])])
|
||||
if min_box_area==0:
|
||||
return 0
|
||||
else:
|
||||
return intersection_area / min_box_area
|
||||
|
||||
|
||||
def get_bbox_in_boundry(bboxes:list, boundry:tuple)-> list:
|
||||
x0, y0, x1, y1 = boundry
|
||||
new_boxes = [box for box in bboxes if box[0] >= x0 and box[1] >= y0 and box[2] <= x1 and box[3] <= y1]
|
||||
return new_boxes
|
||||
|
||||
|
||||
def is_vbox_on_side(bbox, width, height, side_threshold=0.2):
|
||||
"""
|
||||
判断一个bbox是否在pdf页面的边缘
|
||||
"""
|
||||
x0, x1 = bbox[0], bbox[2]
|
||||
if x1<=width*side_threshold or x0>=width*(1-side_threshold):
|
||||
return True
|
||||
return False
|
||||
|
||||
def find_top_nearest_text_bbox(pymu_blocks, obj_bbox):
|
||||
tolerance_margin = 4
|
||||
top_boxes = [box for box in pymu_blocks if obj_bbox[1]-box['bbox'][3] >=-tolerance_margin and not _is_in(box['bbox'], obj_bbox)]
|
||||
# 然后找到X方向上有互相重叠的
|
||||
top_boxes = [box for box in top_boxes if any([obj_bbox[0]-tolerance_margin <=box['bbox'][0]<=obj_bbox[2]+tolerance_margin,
|
||||
obj_bbox[0]-tolerance_margin <=box['bbox'][2]<=obj_bbox[2]+tolerance_margin,
|
||||
box['bbox'][0]-tolerance_margin <=obj_bbox[0]<=box['bbox'][2]+tolerance_margin,
|
||||
box['bbox'][0]-tolerance_margin <=obj_bbox[2]<=box['bbox'][2]+tolerance_margin
|
||||
])]
|
||||
|
||||
# 然后找到y1最大的那个
|
||||
if len(top_boxes)>0:
|
||||
top_boxes.sort(key=lambda x: x['bbox'][3], reverse=True)
|
||||
return top_boxes[0]
|
||||
else:
|
||||
return None
|
||||
|
||||
|
||||
def find_bottom_nearest_text_bbox(pymu_blocks, obj_bbox):
|
||||
bottom_boxes = [box for box in pymu_blocks if box['bbox'][1] - obj_bbox[3]>=-2 and not _is_in(box['bbox'], obj_bbox)]
|
||||
# 然后找到X方向上有互相重叠的
|
||||
bottom_boxes = [box for box in bottom_boxes if any([obj_bbox[0]-2 <=box['bbox'][0]<=obj_bbox[2]+2,
|
||||
obj_bbox[0]-2 <=box['bbox'][2]<=obj_bbox[2]+2,
|
||||
box['bbox'][0]-2 <=obj_bbox[0]<=box['bbox'][2]+2,
|
||||
box['bbox'][0]-2 <=obj_bbox[2]<=box['bbox'][2]+2
|
||||
])]
|
||||
|
||||
# 然后找到y0最小的那个
|
||||
if len(bottom_boxes)>0:
|
||||
bottom_boxes.sort(key=lambda x: x['bbox'][1], reverse=False)
|
||||
return bottom_boxes[0]
|
||||
else:
|
||||
return None
|
||||
|
||||
def find_left_nearest_text_bbox(pymu_blocks, obj_bbox):
|
||||
"""
|
||||
寻找左侧最近的文本block
|
||||
"""
|
||||
left_boxes = [box for box in pymu_blocks if obj_bbox[0]-box['bbox'][2]>=-2 and not _is_in(box['bbox'], obj_bbox)]
|
||||
# 然后找到X方向上有互相重叠的
|
||||
left_boxes = [box for box in left_boxes if any([obj_bbox[1]-2 <=box['bbox'][1]<=obj_bbox[3]+2,
|
||||
obj_bbox[1]-2 <=box['bbox'][3]<=obj_bbox[3]+2,
|
||||
box['bbox'][1]-2 <=obj_bbox[1]<=box['bbox'][3]+2,
|
||||
box['bbox'][1]-2 <=obj_bbox[3]<=box['bbox'][3]+2
|
||||
])]
|
||||
|
||||
# 然后找到x1最大的那个
|
||||
if len(left_boxes)>0:
|
||||
left_boxes.sort(key=lambda x: x['bbox'][2], reverse=True)
|
||||
return left_boxes[0]
|
||||
else:
|
||||
return None
|
||||
|
||||
|
||||
def find_right_nearest_text_bbox(pymu_blocks, obj_bbox):
|
||||
"""
|
||||
寻找右侧最近的文本block
|
||||
"""
|
||||
right_boxes = [box for box in pymu_blocks if box['bbox'][0]-obj_bbox[2]>=-2 and not _is_in(box['bbox'], obj_bbox)]
|
||||
# 然后找到X方向上有互相重叠的
|
||||
right_boxes = [box for box in right_boxes if any([obj_bbox[1]-2 <=box['bbox'][1]<=obj_bbox[3]+2,
|
||||
obj_bbox[1]-2 <=box['bbox'][3]<=obj_bbox[3]+2,
|
||||
box['bbox'][1]-2 <=obj_bbox[1]<=box['bbox'][3]+2,
|
||||
box['bbox'][1]-2 <=obj_bbox[3]<=box['bbox'][3]+2
|
||||
])]
|
||||
|
||||
# 然后找到x0最小的那个
|
||||
if len(right_boxes)>0:
|
||||
right_boxes.sort(key=lambda x: x['bbox'][0], reverse=False)
|
||||
return right_boxes[0]
|
||||
else:
|
||||
return None
|
||||
@@ -0,0 +1,239 @@
|
||||
import os
|
||||
import csv
|
||||
import json
|
||||
import pandas as pd
|
||||
from pandas import DataFrame as df
|
||||
from matplotlib import pyplot as plt
|
||||
from termcolor import cprint
|
||||
|
||||
"""
|
||||
Execute this script in the following way:
|
||||
|
||||
1. Make sure there are pdf_dic.json files under the directory code-clean/tmp/unittest/md/, such as the following:
|
||||
|
||||
code-clean/tmp/unittest/md/scihub/scihub_00500000/libgen.scimag00527000-00527999.zip_10.1002/app.25178/pdf_dic.json
|
||||
|
||||
2. Under the directory code-clean, execute the following command:
|
||||
|
||||
$ python -m libs.calc_span_stats
|
||||
|
||||
"""
|
||||
|
||||
|
||||
def print_green_on_red(text):
|
||||
cprint(text, "green", "on_red", attrs=["bold"], end="\n\n")
|
||||
|
||||
|
||||
def print_green(text):
|
||||
print()
|
||||
cprint(text, "green", attrs=["bold"], end="\n\n")
|
||||
|
||||
|
||||
def print_red(text):
|
||||
print()
|
||||
cprint(text, "red", attrs=["bold"], end="\n\n")
|
||||
|
||||
|
||||
def safe_get(dict_obj, key, default):
|
||||
val = dict_obj.get(key)
|
||||
if val is None:
|
||||
return default
|
||||
else:
|
||||
return val
|
||||
|
||||
|
||||
class SpanStatsCalc:
|
||||
"""Calculate statistics of span."""
|
||||
|
||||
def draw_charts(self, span_stats: pd.DataFrame, fig_num: int, save_path: str):
|
||||
"""Draw multiple figures in one figure."""
|
||||
# make a canvas
|
||||
fig = plt.figure(fig_num, figsize=(20, 20))
|
||||
|
||||
pass
|
||||
|
||||
def calc_stats_per_dict(self, pdf_dict) -> pd.DataFrame:
|
||||
"""Calculate statistics per pdf_dict."""
|
||||
span_stats = pd.DataFrame()
|
||||
|
||||
span_stats = []
|
||||
span_id = 0
|
||||
for page_id, blocks in pdf_dict.items():
|
||||
if page_id.startswith("page_"):
|
||||
if "para_blocks" in blocks.keys():
|
||||
for para_block in blocks["para_blocks"]:
|
||||
for line in para_block["lines"]:
|
||||
for span in line["spans"]:
|
||||
span_text = safe_get(span, "text", "")
|
||||
span_font_name = safe_get(span, "font", "")
|
||||
span_font_size = safe_get(span, "size", 0)
|
||||
span_font_color = safe_get(span, "color", "")
|
||||
span_font_flags = safe_get(span, "flags", 0)
|
||||
|
||||
span_font_flags_decoded = safe_get(span, "decomposed_flags", {})
|
||||
span_is_super_script = safe_get(span_font_flags_decoded, "is_superscript", False)
|
||||
span_is_italic = safe_get(span_font_flags_decoded, "is_italic", False)
|
||||
span_is_serifed = safe_get(span_font_flags_decoded, "is_serifed", False)
|
||||
span_is_sans_serifed = safe_get(span_font_flags_decoded, "is_sans_serifed", False)
|
||||
span_is_monospaced = safe_get(span_font_flags_decoded, "is_monospaced", False)
|
||||
span_is_proportional = safe_get(span_font_flags_decoded, "is_proportional", False)
|
||||
span_is_bold = safe_get(span_font_flags_decoded, "is_bold", False)
|
||||
|
||||
span_stats.append(
|
||||
{
|
||||
"span_id": span_id, # id of span
|
||||
"page_id": page_id, # page number of pdf
|
||||
"span_text": span_text, # text of span
|
||||
"span_font_name": span_font_name, # font name of span
|
||||
"span_font_size": span_font_size, # font size of span
|
||||
"span_font_color": span_font_color, # font color of span
|
||||
"span_font_flags": span_font_flags, # font flags of span
|
||||
"span_is_superscript": int(
|
||||
span_is_super_script
|
||||
), # indicate whether the span is super script or not
|
||||
"span_is_italic": int(span_is_italic), # indicate whether the span is italic or not
|
||||
"span_is_serifed": int(span_is_serifed), # indicate whether the span is serifed or not
|
||||
"span_is_sans_serifed": int(
|
||||
span_is_sans_serifed
|
||||
), # indicate whether the span is sans serifed or not
|
||||
"span_is_monospaced": int(
|
||||
span_is_monospaced
|
||||
), # indicate whether the span is monospaced or not
|
||||
"span_is_proportional": int(
|
||||
span_is_proportional
|
||||
), # indicate whether the span is proportional or not
|
||||
"span_is_bold": int(span_is_bold), # indicate whether the span is bold or not
|
||||
}
|
||||
)
|
||||
|
||||
span_id += 1
|
||||
|
||||
span_stats = pd.DataFrame(span_stats)
|
||||
# print(span_stats)
|
||||
|
||||
return span_stats
|
||||
|
||||
|
||||
def __find_pdf_dic_files(
|
||||
jf_name="pdf_dic.json",
|
||||
base_code_name="code-clean",
|
||||
tgt_base_dir_name="tmp",
|
||||
unittest_dir_name="unittest",
|
||||
md_dir_name="md",
|
||||
book_names=[
|
||||
"scihub",
|
||||
], # other possible values: "zlib", "arxiv" and so on
|
||||
):
|
||||
pdf_dict_files = []
|
||||
|
||||
curr_dir = os.path.dirname(__file__)
|
||||
|
||||
for i in range(len(curr_dir)):
|
||||
if curr_dir[i : i + len(base_code_name)] == base_code_name:
|
||||
base_code_dir_name = curr_dir[: i + len(base_code_name)]
|
||||
for book_name in book_names:
|
||||
search_dir_relative_name = os.path.join(tgt_base_dir_name, unittest_dir_name, md_dir_name, book_name)
|
||||
if os.path.exists(base_code_dir_name):
|
||||
search_dir_name = os.path.join(base_code_dir_name, search_dir_relative_name)
|
||||
for root, dirs, files in os.walk(search_dir_name):
|
||||
for file in files:
|
||||
if file == jf_name:
|
||||
pdf_dict_files.append(os.path.join(root, file))
|
||||
break
|
||||
|
||||
return pdf_dict_files
|
||||
|
||||
|
||||
def combine_span_texts(group_df, span_stats):
|
||||
combined_span_texts = []
|
||||
for _, row in group_df.iterrows():
|
||||
curr_span_id = row.name
|
||||
curr_span_text = row["span_text"]
|
||||
|
||||
pre_span_id = curr_span_id - 1
|
||||
pre_span_text = span_stats.at[pre_span_id, "span_text"] if pre_span_id in span_stats.index else ""
|
||||
|
||||
next_span_id = curr_span_id + 1
|
||||
next_span_text = span_stats.at[next_span_id, "span_text"] if next_span_id in span_stats.index else ""
|
||||
|
||||
# pointer_sign is a right arrow if the span is superscript, otherwise it is a down arrow
|
||||
pointer_sign = "→ → → "
|
||||
combined_text = "\n".join([pointer_sign + pre_span_text, pointer_sign + curr_span_text, pointer_sign + next_span_text])
|
||||
combined_span_texts.append(combined_text)
|
||||
|
||||
return "\n\n".join(combined_span_texts)
|
||||
|
||||
|
||||
# pd.set_option("display.max_colwidth", None) # 设置为 None 来显示完整的文本
|
||||
pd.set_option("display.max_rows", None) # 设置为 None 来显示更多的行
|
||||
|
||||
|
||||
def main():
|
||||
pdf_dict_files = __find_pdf_dic_files()
|
||||
# print(pdf_dict_files)
|
||||
|
||||
span_stats_calc = SpanStatsCalc()
|
||||
|
||||
for pdf_dict_file in pdf_dict_files:
|
||||
print("-" * 100)
|
||||
print_green_on_red(f"Processing {pdf_dict_file}")
|
||||
|
||||
with open(pdf_dict_file, "r", encoding="utf-8") as f:
|
||||
pdf_dict = json.load(f)
|
||||
|
||||
raw_df = span_stats_calc.calc_stats_per_dict(pdf_dict)
|
||||
save_path = pdf_dict_file.replace("pdf_dic.json", "span_stats_raw.csv")
|
||||
raw_df.to_csv(save_path, index=False)
|
||||
|
||||
filtered_df = raw_df[raw_df["span_is_superscript"] == 1]
|
||||
if filtered_df.empty:
|
||||
print("No superscript span found!")
|
||||
continue
|
||||
|
||||
filtered_grouped_df = filtered_df.groupby(["span_font_name", "span_font_size", "span_font_color"])
|
||||
|
||||
combined_span_texts = filtered_grouped_df.apply(combine_span_texts, span_stats=raw_df) # type: ignore
|
||||
|
||||
final_df = filtered_grouped_df.size().reset_index(name="count")
|
||||
final_df["span_texts"] = combined_span_texts.reset_index(level=[0, 1, 2], drop=True)
|
||||
|
||||
print(final_df)
|
||||
|
||||
final_df["span_texts"] = final_df["span_texts"].apply(lambda x: x.replace("\n", "\r\n"))
|
||||
|
||||
save_path = pdf_dict_file.replace("pdf_dic.json", "span_stats_final.csv")
|
||||
# 使用 UTF-8 编码并添加 BOM,确保所有字段被双引号包围
|
||||
final_df.to_csv(save_path, index=False, encoding="utf-8-sig", quoting=csv.QUOTE_ALL)
|
||||
|
||||
# 创建一个 2x2 的图表布局
|
||||
fig, axs = plt.subplots(2, 2, figsize=(15, 10))
|
||||
|
||||
# 按照 span_font_name 分类作图
|
||||
final_df.groupby("span_font_name")["count"].sum().plot(kind="bar", ax=axs[0, 0], title="By Font Name")
|
||||
|
||||
# 按照 span_font_size 分类作图
|
||||
final_df.groupby("span_font_size")["count"].sum().plot(kind="bar", ax=axs[0, 1], title="By Font Size")
|
||||
|
||||
# 按照 span_font_color 分类作图
|
||||
final_df.groupby("span_font_color")["count"].sum().plot(kind="bar", ax=axs[1, 0], title="By Font Color")
|
||||
|
||||
# 按照 span_font_name、span_font_size 和 span_font_color 共同分类作图
|
||||
grouped = final_df.groupby(["span_font_name", "span_font_size", "span_font_color"])
|
||||
grouped["count"].sum().unstack().plot(kind="bar", ax=axs[1, 1], title="Combined Grouping")
|
||||
|
||||
# 调整布局
|
||||
plt.tight_layout()
|
||||
|
||||
# 显示图表
|
||||
# plt.show()
|
||||
|
||||
# 保存图表到 PNG 文件
|
||||
save_path = pdf_dict_file.replace("pdf_dic.json", "span_stats_combined.png")
|
||||
plt.savefig(save_path)
|
||||
|
||||
# 清除画布
|
||||
plt.clf()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
+213
@@ -0,0 +1,213 @@
|
||||
import datetime
|
||||
import os, re, configparser
|
||||
import time
|
||||
|
||||
import boto3
|
||||
from loguru import logger
|
||||
from boto3.s3.transfer import TransferConfig
|
||||
from botocore.config import Config
|
||||
|
||||
import fitz # 1.23.9中已经切换到rebase
|
||||
# import fitz_new as fitz # 使用rebased的新版pymupdf库
|
||||
|
||||
def get_delta_time(input_time):
|
||||
return round(time.time() - input_time, 2)
|
||||
|
||||
|
||||
def join_path(*args):
|
||||
return '/'.join(s.rstrip('/') for s in args)
|
||||
|
||||
#配置全局的errlog_path,方便demo同步引用
|
||||
error_log_path = "s3://llm-pdf-text/err_logs/"
|
||||
# json_dump_path = "s3://pdf_books_temp/json_dump/" # 这条路径仅用于临时本地测试,不能提交到main
|
||||
json_dump_path = "s3://llm-pdf-text/json_dump/"
|
||||
|
||||
|
||||
def get_top_percent_list(num_list, percent):
|
||||
"""
|
||||
获取列表中前百分之多少的元素
|
||||
:param num_list:
|
||||
:param percent:
|
||||
:return:
|
||||
"""
|
||||
if len(num_list) == 0:
|
||||
top_percent_list = []
|
||||
else:
|
||||
# 对imgs_len_list排序
|
||||
sorted_imgs_len_list = sorted(num_list, reverse=True)
|
||||
# 计算 percent 的索引
|
||||
top_percent_index = int(len(sorted_imgs_len_list) * percent)
|
||||
# 取前80%的元素
|
||||
top_percent_list = sorted_imgs_len_list[:top_percent_index]
|
||||
return top_percent_list
|
||||
|
||||
|
||||
def formatted_time(time_stamp):
|
||||
dt_object = datetime.datetime.fromtimestamp(time_stamp)
|
||||
output_time = dt_object.strftime("%Y-%m-%d-%H:%M:%S")
|
||||
return output_time
|
||||
|
||||
|
||||
def mymax(alist: list):
|
||||
if len(alist) == 0:
|
||||
return 0 # 空是0, 0*0也是0大小q
|
||||
else:
|
||||
return max(alist)
|
||||
|
||||
def parse_aws_param(profile):
|
||||
if isinstance(profile, str):
|
||||
# 解析配置文件
|
||||
config_file = join_path(os.path.expanduser("~"), ".aws", "config")
|
||||
credentials_file = join_path(os.path.expanduser("~"), ".aws", "credentials")
|
||||
config = configparser.ConfigParser()
|
||||
config.read(credentials_file)
|
||||
config.read(config_file)
|
||||
# 获取 AWS 账户相关信息
|
||||
ak = config.get(profile, "aws_access_key_id")
|
||||
sk = config.get(profile, "aws_secret_access_key")
|
||||
if profile == "default":
|
||||
s3_str = config.get(f"{profile}", "s3")
|
||||
else:
|
||||
s3_str = config.get(f"profile {profile}", "s3")
|
||||
end_match = re.search("endpoint_url[\s]*=[\s]*([^\s\n]+)[\s\n]*$", s3_str, re.MULTILINE)
|
||||
if end_match:
|
||||
endpoint = end_match.group(1)
|
||||
else:
|
||||
raise ValueError(f"aws 配置文件中没有找到 endpoint_url")
|
||||
style_match = re.search("addressing_style[\s]*=[\s]*([^\s\n]+)[\s\n]*$", s3_str, re.MULTILINE)
|
||||
if style_match:
|
||||
addressing_style = style_match.group(1)
|
||||
else:
|
||||
addressing_style = "path"
|
||||
elif isinstance(profile, dict):
|
||||
ak = profile["ak"]
|
||||
sk = profile["sk"]
|
||||
endpoint = profile["endpoint"]
|
||||
addressing_style = "auto"
|
||||
|
||||
return ak, sk, endpoint, addressing_style
|
||||
|
||||
|
||||
def parse_bucket_key(s3_full_path: str):
|
||||
"""
|
||||
输入 s3://bucket/path/to/my/file.txt
|
||||
输出 bucket, path/to/my/file.txt
|
||||
"""
|
||||
s3_full_path = s3_full_path.strip()
|
||||
if s3_full_path.startswith("s3://"):
|
||||
s3_full_path = s3_full_path[5:]
|
||||
if s3_full_path.startswith("/"):
|
||||
s3_full_path = s3_full_path[1:]
|
||||
bucket, key = s3_full_path.split("/", 1)
|
||||
return bucket, key
|
||||
|
||||
|
||||
def read_file(pdf_path: str, s3_profile):
|
||||
if pdf_path.startswith("s3://"):
|
||||
ak, sk, end_point, addressing_style = parse_aws_param(s3_profile)
|
||||
cli = boto3.client(service_name="s3", aws_access_key_id=ak, aws_secret_access_key=sk, endpoint_url=end_point,
|
||||
config=Config(s3={'addressing_style': addressing_style}, retries={'max_attempts': 10, 'mode': 'standard'}))
|
||||
bucket_name, bucket_key = parse_bucket_key(pdf_path)
|
||||
res = cli.get_object(Bucket=bucket_name, Key=bucket_key)
|
||||
file_content = res["Body"].read()
|
||||
return file_content
|
||||
else:
|
||||
with open(pdf_path, "rb") as f:
|
||||
return f.read()
|
||||
|
||||
def list_dir(dir_path:str, s3_profile:str):
|
||||
"""
|
||||
列出dir_path下的所有文件
|
||||
"""
|
||||
ret = []
|
||||
|
||||
if dir_path.startswith("s3"):
|
||||
ak, sk, end_point, addressing_style = parse_aws_param(s3_profile)
|
||||
s3info = re.findall(r"s3:\/\/([^\/]+)\/(.*)", dir_path)
|
||||
bucket, path = s3info[0][0], s3info[0][1]
|
||||
try:
|
||||
cli = boto3.client(service_name="s3", aws_access_key_id=ak, aws_secret_access_key=sk, endpoint_url=end_point,
|
||||
config=Config(s3={'addressing_style': addressing_style}))
|
||||
def list_obj_scluster():
|
||||
marker = None
|
||||
while True:
|
||||
list_kwargs = dict(MaxKeys=1000, Bucket=bucket, Prefix=path)
|
||||
if marker:
|
||||
list_kwargs['Marker'] = marker
|
||||
response = cli.list_objects(**list_kwargs)
|
||||
contents = response.get("Contents", [])
|
||||
yield from contents
|
||||
if not response.get("IsTruncated") or len(contents)==0:
|
||||
break
|
||||
marker = contents[-1]['Key']
|
||||
|
||||
|
||||
for info in list_obj_scluster():
|
||||
file_path = info['Key']
|
||||
#size = info['Size']
|
||||
|
||||
if path!="":
|
||||
afile = file_path[len(path):]
|
||||
if afile.endswith(".json"):
|
||||
ret.append(f"s3://{bucket}/{file_path}")
|
||||
|
||||
return ret
|
||||
|
||||
except Exception as e:
|
||||
logger.exception(e)
|
||||
exit(-1)
|
||||
else: #本地的目录,那么扫描本地目录并返会这个目录里的所有jsonl文件
|
||||
|
||||
for root, dirs, files in os.walk(dir_path):
|
||||
for file in files:
|
||||
if file.endswith(".json"):
|
||||
ret.append(join_path(root, file))
|
||||
ret.sort()
|
||||
return ret
|
||||
|
||||
def get_img_s3_client(save_path:str, image_s3_config:str):
|
||||
"""
|
||||
"""
|
||||
if save_path.startswith("s3://"): # 放这里是为了最少创建一个s3 client
|
||||
ak, sk, end_point, addressing_style = parse_aws_param(image_s3_config)
|
||||
img_s3_client = boto3.client(
|
||||
service_name="s3",
|
||||
aws_access_key_id=ak,
|
||||
aws_secret_access_key=sk,
|
||||
endpoint_url=end_point,
|
||||
config=Config(s3={"addressing_style": addressing_style}, retries={'max_attempts': 5, 'mode': 'standard'}),
|
||||
)
|
||||
else:
|
||||
img_s3_client = None
|
||||
|
||||
return img_s3_client
|
||||
|
||||
# def get_s3_object(path):
|
||||
# src_cli_config = Config(**{
|
||||
#
|
||||
# "connect_timeout": 60,
|
||||
# "read_timeout": 20,
|
||||
# "max_pool_connections": 500,
|
||||
# "s3": {
|
||||
# "addressing_style": "path",
|
||||
# },
|
||||
# "retries": {
|
||||
# "max_attempts": 3,
|
||||
# }
|
||||
# })
|
||||
# full_path = f"{bucket_name}/{bucket_prefix}/{path}"
|
||||
# try:
|
||||
# src_cli = boto3.session.Session().client("s3", aws_access_key_id=ak, aws_secret_access_key=sk, endpoint_url=endpoint, region_name='', config=src_cli_config)
|
||||
# res = src_cli.get_object(Bucket=bucket_name, Key=f"{bucket_prefix}/{path}")
|
||||
# file_content = res["Body"].read()
|
||||
# return file_content
|
||||
# except Exception as e:
|
||||
# logger.error(f"get_s3_object({full_path}) error: {e}")
|
||||
# return b''
|
||||
|
||||
if __name__=="__main__":
|
||||
s3_path = "s3://llm-pdf-text/layout_det/scihub/scimag07865000-07865999/10.1007/s10729-011-9175-6.pdf/"
|
||||
s3_profile = "langchao"
|
||||
ret = list_dir(s3_path, s3_profile)
|
||||
print(ret)
|
||||
|
||||
@@ -0,0 +1,24 @@
|
||||
|
||||
class DropReason:
|
||||
TEXT_BLCOK_HOR_OVERLAP = "text_block_horizontal_overlap" # 文字块有水平互相覆盖,导致无法准确定位文字顺序
|
||||
COMPLICATED_LAYOUT = "complicated_layout" # 复杂的布局,暂时不支持
|
||||
TOO_MANY_LAYOUT_COLUMNS = "too_many_layout_columns" # 目前不支持分栏超过2列的
|
||||
COLOR_BACKGROUND_TEXT_BOX = "color_background_text_box" # 含有带色块的PDF,色块会改变阅读顺序,目前不支持带底色文字块的PDF。
|
||||
HIGH_COMPUTATIONAL_lOAD_BY_IMGS = "high_computational_load_by_imgs" # 含特殊图片,计算量太大,从而丢弃
|
||||
HIGH_COMPUTATIONAL_lOAD_BY_SVGS = "high_computational_load_by_svgs" # 特殊的SVG图,计算量太大,从而丢弃
|
||||
HIGH_COMPUTATIONAL_lOAD_BY_TOTAL_PAGES = "high_computational_load_by_total_pages" # 计算量超过负荷,当前方法下计算量消耗过大
|
||||
MISS_DOC_LAYOUT_RESULT = "missing doc_layout_result" # 版面分析失败
|
||||
Exception = "exception" # 解析中发生异常
|
||||
ENCRYPTED = "encrypted" # PDF是加密的
|
||||
EMPTY_PDF = "total_page=0" # PDF页面总数为0
|
||||
NOT_IS_TEXT_PDF = "not_is_text_pdf" # 不是文字版PDF,无法直接解析
|
||||
DENSE_SINGLE_LINE_BLOCK = "dense_single_line_block" # 无法清晰的分段
|
||||
TITLE_DETECTION_FAILED = "title_detection_failed" # 探测标题失败
|
||||
TITLE_LEVEL_FAILED = "title_level_failed" # 分析标题级别失败(例如一级、二级、三级标题)
|
||||
PARA_SPLIT_FAILED = "para_split_failed" # 识别段落失败
|
||||
PARA_MERGE_FAILED = "para_merge_failed" # 段落合并失败
|
||||
NOT_ALLOW_LANGUAGE = "not_allow_language" # 不支持的语种
|
||||
SPECIAL_PDF = "special_pdf"
|
||||
PSEUDO_SINGLE_COLUMN = "pseudo_single_column" # 无法精确判断文字分栏
|
||||
CAN_NOT_DETECT_PAGE_LAYOUT="can_not_detect_page_layout" # 无法分析页面的版面
|
||||
|
||||
@@ -0,0 +1,2 @@
|
||||
|
||||
COLOR_BG_HEADER_TXT_BLOCK = "color_background_header_txt_block"
|
||||
@@ -0,0 +1,27 @@
|
||||
import json
|
||||
import brotli
|
||||
import base64
|
||||
|
||||
class JsonCompressor:
|
||||
|
||||
@staticmethod
|
||||
def compress_json(data):
|
||||
"""
|
||||
Compress a json object and encode it with base64
|
||||
"""
|
||||
json_str = json.dumps(data)
|
||||
json_bytes = json_str.encode('utf-8')
|
||||
compressed = brotli.compress(json_bytes, quality=6)
|
||||
compressed_str = base64.b64encode(compressed).decode('utf-8') # convert bytes to string
|
||||
return compressed_str
|
||||
|
||||
@staticmethod
|
||||
def decompress_json(compressed_str):
|
||||
"""
|
||||
Decode the base64 string and decompress the json object
|
||||
"""
|
||||
compressed = base64.b64decode(compressed_str.encode('utf-8')) # convert string to bytes
|
||||
decompressed_bytes = brotli.decompress(compressed)
|
||||
json_str = decompressed_bytes.decode('utf-8')
|
||||
data = json.loads(json_str)
|
||||
return data
|
||||
@@ -0,0 +1,36 @@
|
||||
import pycld2 as cld2
|
||||
import regex
|
||||
import unicodedata
|
||||
|
||||
|
||||
RE_BAD_CHARS = regex.compile(r"\p{Cc}|\p{Cs}")
|
||||
|
||||
|
||||
def remove_bad_chars(text):
|
||||
return RE_BAD_CHARS.sub("", text)
|
||||
|
||||
|
||||
def detect_lang(text: str) -> str:
|
||||
if len(text) == 0:
|
||||
return ""
|
||||
|
||||
try:
|
||||
_, _, details = cld2.detect(text)
|
||||
except:
|
||||
# cld2 doesn't like control characters
|
||||
# https://github.com/mikemccand/chromium-compact-language-detector/issues/22#issuecomment-435904616
|
||||
html_no_ctrl_chars = ''.join([l for l in text if unicodedata.category(l)[0] not in ['C',]])
|
||||
_, _, details = cld2.detect(html_no_ctrl_chars)
|
||||
lang = ""
|
||||
try:
|
||||
lang = details[0][1].lower()
|
||||
except:
|
||||
lang = ""
|
||||
return lang
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
print(detect_lang("This is a test."))
|
||||
print(detect_lang("<html>This is a test</html>"))
|
||||
print(detect_lang("这个是中文测试。"))
|
||||
print(detect_lang("<html>这个是中文测试。</html>"))
|
||||
@@ -0,0 +1,20 @@
|
||||
import re
|
||||
|
||||
|
||||
def escape_special_markdown_char(pymu_blocks):
|
||||
"""
|
||||
转义正文里对markdown语法有特殊意义的字符
|
||||
"""
|
||||
special_chars = ["*", "`", "~", "$"]
|
||||
for blk in pymu_blocks:
|
||||
for line in blk['lines']:
|
||||
for span in line['spans']:
|
||||
for char in special_chars:
|
||||
span_text = span['text']
|
||||
span_type = span.get("_type", None)
|
||||
if span_type in ['inline-equation', 'interline-equation']:
|
||||
continue
|
||||
elif span_text:
|
||||
span['text'] = span['text'].replace(char, "\\" + char)
|
||||
|
||||
return pymu_blocks
|
||||
@@ -0,0 +1,203 @@
|
||||
import re
|
||||
from os import path
|
||||
|
||||
from collections import Counter
|
||||
|
||||
from loguru import logger
|
||||
|
||||
# from langdetect import detect
|
||||
import spacy
|
||||
import en_core_web_sm
|
||||
import zh_core_web_sm
|
||||
|
||||
from libs.language import detect_lang
|
||||
|
||||
|
||||
class NLPModels:
|
||||
"""
|
||||
How to upload local models to s3:
|
||||
- config aws cli:
|
||||
doc\SETUP-CLI.md
|
||||
doc\setup_cli.sh
|
||||
app\config\__init__.py
|
||||
- $ cd {local_dir_storing_models}
|
||||
- $ ls models
|
||||
en_core_web_sm-3.7.1/
|
||||
zh_core_web_sm-3.7.0/
|
||||
- $ aws s3 sync models/ s3://llm-infra/models --profile=p_project_norm
|
||||
- $ aws s3 --profile=p_project_norm ls s3://llm-infra/models/
|
||||
PRE en_core_web_sm-3.7.1/
|
||||
PRE zh_core_web_sm-3.7.0/
|
||||
"""
|
||||
|
||||
def __init__(self):
|
||||
# if OS is windows, set "TMP_DIR" to "D:/tmp"
|
||||
|
||||
home_dir = path.expanduser("~")
|
||||
self.default_local_path = path.join(home_dir, ".nlp_models")
|
||||
self.default_shared_path = "/share/pdf_processor/nlp_models"
|
||||
self.default_hdfs_path = "hdfs://pdf_processor/nlp_models"
|
||||
self.default_s3_path = "s3://llm-infra/models"
|
||||
self.nlp_models = self.nlp_models = {
|
||||
"en_core_web_sm": {
|
||||
"type": "spacy",
|
||||
"version": "3.7.1",
|
||||
},
|
||||
"en_core_web_md": {
|
||||
"type": "spacy",
|
||||
"version": "3.7.1",
|
||||
},
|
||||
"en_core_web_lg": {
|
||||
"type": "spacy",
|
||||
"version": "3.7.1",
|
||||
},
|
||||
"zh_core_web_sm": {
|
||||
"type": "spacy",
|
||||
"version": "3.7.0",
|
||||
},
|
||||
"zh_core_web_md": {
|
||||
"type": "spacy",
|
||||
"version": "3.7.0",
|
||||
},
|
||||
"zh_core_web_lg": {
|
||||
"type": "spacy",
|
||||
"version": "3.7.0",
|
||||
},
|
||||
}
|
||||
self.en_core_web_sm_model = en_core_web_sm.load()
|
||||
self.zh_core_web_sm_model = zh_core_web_sm.load()
|
||||
|
||||
def load_model(self, model_name, model_type, model_version):
|
||||
if (
|
||||
model_name in self.nlp_models
|
||||
and self.nlp_models[model_name]["type"] == model_type
|
||||
and self.nlp_models[model_name]["version"] == model_version
|
||||
):
|
||||
return spacy.load(model_name) if spacy.util.is_package(model_name) else None
|
||||
|
||||
else:
|
||||
logger.error(f"Unsupported model name or version: {model_name} {model_version}")
|
||||
return None
|
||||
|
||||
def detect_language(self, text, use_langdetect=False):
|
||||
if len(text) == 0:
|
||||
return None
|
||||
if use_langdetect:
|
||||
# print("use_langdetect")
|
||||
# print(detect_lang(text))
|
||||
# return detect_lang(text)
|
||||
if detect_lang(text) == "zh":
|
||||
return "zh"
|
||||
else:
|
||||
return "en"
|
||||
|
||||
if not use_langdetect:
|
||||
en_count = len(re.findall(r"[a-zA-Z]", text))
|
||||
cn_count = len(re.findall(r"[\u4e00-\u9fff]", text))
|
||||
|
||||
if en_count > cn_count:
|
||||
return "en"
|
||||
|
||||
if cn_count > en_count:
|
||||
return "zh"
|
||||
|
||||
def detect_entity_catgr_using_nlp(self, text, threshold=0.5):
|
||||
"""
|
||||
Detect entity categories using NLP models and return the most frequent entity types.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
text : str
|
||||
Text to be processed.
|
||||
|
||||
Returns
|
||||
-------
|
||||
str
|
||||
The most frequent entity type.
|
||||
"""
|
||||
lang = self.detect_language(text, use_langdetect=True)
|
||||
|
||||
if lang == "en":
|
||||
nlp_model = self.en_core_web_sm_model
|
||||
elif lang == "zh":
|
||||
nlp_model = self.zh_core_web_sm_model
|
||||
else:
|
||||
# logger.error(f"Unsupported language: {lang}")
|
||||
return {}
|
||||
|
||||
# Splitting text into smaller parts
|
||||
text_parts = re.split(r"[,;,;、\s & |]+", text)
|
||||
|
||||
text_parts = [part for part in text_parts if not re.match(r"[\d\W]+", part)] # Remove non-words
|
||||
text_combined = " ".join(text_parts)
|
||||
|
||||
try:
|
||||
doc = nlp_model(text_combined)
|
||||
entity_counts = Counter([ent.label_ for ent in doc.ents])
|
||||
word_counts_in_entities = Counter()
|
||||
|
||||
for ent in doc.ents:
|
||||
word_counts_in_entities[ent.label_] += len(ent.text.split())
|
||||
|
||||
total_words_in_entities = sum(word_counts_in_entities.values())
|
||||
total_words = len([token for token in doc if not token.is_punct])
|
||||
|
||||
if total_words_in_entities == 0 or total_words == 0:
|
||||
return None
|
||||
|
||||
entity_percentage = total_words_in_entities / total_words
|
||||
if entity_percentage < 0.5:
|
||||
return None
|
||||
|
||||
most_common_entity, word_count = word_counts_in_entities.most_common(1)[0]
|
||||
entity_percentage = word_count / total_words_in_entities
|
||||
|
||||
if entity_percentage >= threshold:
|
||||
return most_common_entity
|
||||
else:
|
||||
return None
|
||||
except Exception as e:
|
||||
logger.error(f"Error in entity detection: {e}")
|
||||
return None
|
||||
|
||||
|
||||
def __main__():
|
||||
nlpModel = NLPModels()
|
||||
|
||||
test_strings = [
|
||||
"张三",
|
||||
"张三, 李四,王五; 赵六",
|
||||
"John Doe",
|
||||
"Jane Smith",
|
||||
"Lee, John",
|
||||
"John Doe, Jane Smith; Alice Johnson,Bob Lee",
|
||||
"孙七, Michael Jordan;赵八",
|
||||
"David Smith Michael O'Connor; Kevin ßáçøñ",
|
||||
"李雷·韩梅梅, 张三·李四",
|
||||
"Charles Robert Darwin, Isaac Newton",
|
||||
"莱昂纳多·迪卡普里奥, 杰克·吉伦哈尔",
|
||||
"John Doe, Jane Smith; Alice Johnson",
|
||||
"张三, 李四,王五; 赵六",
|
||||
"Lei Wang, Jia Li, and Xiaojun Chen, LINKE YANG OU, and YUAN ZHANG",
|
||||
"Rachel Mills & William Barry & Susanne B. Haga",
|
||||
"Claire Chabut* and Jean-François Bussières",
|
||||
"1 Department of Chemistry, Northeastern University, Shenyang 110004, China 2 State Key Laboratory of Polymer Physics and Chemistry, Changchun Institute of Applied Chemistry, Chinese Academy of Sciences, Changchun 130022, China",
|
||||
"Changchun",
|
||||
"china",
|
||||
"Rongjun Song, 1,2 Baoyan Zhang, 1 Baotong Huang, 2 Tao Tang 2",
|
||||
"Synergistic Effect of Supported Nickel Catalyst with Intumescent Flame-Retardants on Flame Retardancy and Thermal Stability of Polypropylene",
|
||||
"Synergistic Effect of Supported Nickel Catalyst with",
|
||||
"Intumescent Flame-Retardants on Flame Retardancy",
|
||||
"and Thermal Stability of Polypropylene",
|
||||
]
|
||||
|
||||
for test in test_strings:
|
||||
print()
|
||||
print(f"Original String: {test}")
|
||||
|
||||
result = nlpModel.detect_entity_catgr_using_nlp(test)
|
||||
print(f"Detected entities: {result}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
__main__()
|
||||
@@ -0,0 +1,137 @@
|
||||
import os
|
||||
from pathlib import Path
|
||||
from typing import Tuple
|
||||
import io
|
||||
|
||||
# from app.common.s3 import get_s3_client
|
||||
from libs.commons import fitz
|
||||
from loguru import logger
|
||||
from libs.commons import parse_bucket_key, join_path
|
||||
|
||||
|
||||
def cut_image(bbox: Tuple, page_num: int, page: fitz.Page, save_parent_path: str, s3_return_path=None, img_s3_client=None, upload_switch=True):
|
||||
"""
|
||||
从第page_num页的page中,根据bbox进行裁剪出一张jpg图片,返回图片路径
|
||||
save_path:需要同时支持s3和本地, 图片存放在save_path下,文件名是: {page_num}_{bbox[0]}_{bbox[1]}_{bbox[2]}_{bbox[3]}.jpg , bbox内数字取整。
|
||||
"""
|
||||
# 拼接文件名
|
||||
filename = f"{page_num}_{int(bbox[0])}_{int(bbox[1])}_{int(bbox[2])}_{int(bbox[3])}.jpg"
|
||||
# 拼接路径
|
||||
image_save_path = join_path(save_parent_path, filename)
|
||||
s3_img_path = join_path(s3_return_path, filename) if s3_return_path is not None else None
|
||||
# 打印图片文件名
|
||||
# print(f"Saved {image_save_path}")
|
||||
|
||||
#检查坐标
|
||||
# x_check = int(bbox[2]) - int(bbox[0])
|
||||
# y_check = int(bbox[3]) - int(bbox[1])
|
||||
# if x_check <= 0 or y_check <= 0:
|
||||
#
|
||||
# if image_save_path.startswith("s3://"):
|
||||
# logger.exception(f"传入图片坐标有误,x1<x0或y1<y0,{s3_img_path}")
|
||||
# return s3_img_path
|
||||
# else:
|
||||
# logger.exception(f"传入图片坐标有误,x1<x0或y1<y0,{image_save_path}")
|
||||
# return image_save_path
|
||||
|
||||
|
||||
# 将坐标转换为fitz.Rect对象
|
||||
rect = fitz.Rect(*bbox)
|
||||
# 配置缩放倍数为3倍
|
||||
zoom = fitz.Matrix(3, 3)
|
||||
# 截取图片
|
||||
pix = page.get_pixmap(clip=rect, matrix=zoom)
|
||||
|
||||
if image_save_path.startswith("s3://"):
|
||||
if not upload_switch:
|
||||
pass
|
||||
else:
|
||||
# 图片保存到s3
|
||||
bucket_name, bucket_key = parse_bucket_key(image_save_path)
|
||||
# 将字节流上传到s3
|
||||
byte_data = pix.tobytes(output='jpeg', jpg_quality=95)
|
||||
file_obj = io.BytesIO(byte_data)
|
||||
if img_s3_client is not None:
|
||||
img_s3_client.upload_fileobj(file_obj, bucket_name, bucket_key)
|
||||
# 每个图片上传任务都创建一个新的client
|
||||
# img_s3_client_once = get_s3_client(image_save_path)
|
||||
# img_s3_client_once.upload_fileobj(file_obj, bucket_name, bucket_key)
|
||||
else:
|
||||
logger.exception("must input img_s3_client")
|
||||
return s3_img_path
|
||||
else:
|
||||
# 保存图片到本地
|
||||
# 先检查一下image_save_path的父目录是否存在,如果不存在,就创建
|
||||
parent_dir = os.path.dirname(image_save_path)
|
||||
if not os.path.exists(parent_dir):
|
||||
os.makedirs(parent_dir)
|
||||
pix.save(image_save_path, jpg_quality=95)
|
||||
# 为了直接能在markdown里看,这里把地址改为相对于mardown的地址
|
||||
pth = Path(image_save_path)
|
||||
image_save_path = f"{pth.parent.name}/{pth.name}"
|
||||
return image_save_path
|
||||
|
||||
|
||||
def save_images_by_bboxes(book_name: str, page_num: int, page: fitz.Page, save_path: str,
|
||||
image_bboxes: list, images_overlap_backup:list, table_bboxes: list, equation_inline_bboxes: list,
|
||||
equation_interline_bboxes: list, img_s3_client) -> dict:
|
||||
"""
|
||||
返回一个dict, key为bbox, 值是图片地址
|
||||
"""
|
||||
image_info = []
|
||||
image_backup_info = []
|
||||
table_info = []
|
||||
inline_eq_info = []
|
||||
interline_eq_info = []
|
||||
|
||||
# 图片的保存路径组成是这样的: {s3_or_local_path}/{book_name}/{images|tables|equations}/{page_num}_{bbox[0]}_{bbox[1]}_{bbox[2]}_{bbox[3]}.jpg
|
||||
s3_return_image_path = join_path(book_name, "images")
|
||||
image_save_path = join_path(save_path, s3_return_image_path)
|
||||
|
||||
s3_return_table_path = join_path(book_name, "tables")
|
||||
table_save_path = join_path(save_path, s3_return_table_path)
|
||||
|
||||
s3_return_equations_inline_path = join_path(book_name, "equations_inline")
|
||||
equation_inline_save_path = join_path(save_path, s3_return_equations_inline_path)
|
||||
|
||||
s3_return_equation_interline_path = join_path(book_name, "equation_interline")
|
||||
equation_interline_save_path = join_path(save_path, s3_return_equation_interline_path)
|
||||
|
||||
|
||||
for bbox in image_bboxes:
|
||||
if any([bbox[0]>=bbox[2], bbox[1]>=bbox[3]]):
|
||||
logger.warning(f"image_bboxes: 错误的box, {bbox}")
|
||||
continue
|
||||
|
||||
image_path = cut_image(bbox, page_num, page, image_save_path, s3_return_image_path, img_s3_client)
|
||||
image_info.append({"bbox": bbox, "image_path": image_path})
|
||||
|
||||
for bbox in images_overlap_backup:
|
||||
if any([bbox[0]>=bbox[2], bbox[1]>=bbox[3]]):
|
||||
logger.warning(f"images_overlap_backup: 错误的box, {bbox}")
|
||||
continue
|
||||
image_path = cut_image(bbox, page_num, page, image_save_path, s3_return_image_path, img_s3_client)
|
||||
image_backup_info.append({"bbox": bbox, "image_path": image_path})
|
||||
|
||||
for bbox in table_bboxes:
|
||||
if any([bbox[0]>=bbox[2], bbox[1]>=bbox[3]]):
|
||||
logger.warning(f"table_bboxes: 错误的box, {bbox}")
|
||||
continue
|
||||
image_path = cut_image(bbox, page_num, page, table_save_path, s3_return_table_path, img_s3_client)
|
||||
table_info.append({"bbox": bbox, "image_path": image_path})
|
||||
|
||||
for bbox in equation_inline_bboxes:
|
||||
if any([bbox[0]>=bbox[2], bbox[1]>=bbox[3]]):
|
||||
logger.warning(f"equation_inline_bboxes: 错误的box, {bbox}")
|
||||
continue
|
||||
image_path = cut_image(bbox[:4], page_num, page, equation_inline_save_path, s3_return_equations_inline_path, img_s3_client, upload_switch=False)
|
||||
inline_eq_info.append({'bbox':bbox[:4], "image_path":image_path, "latex_text":bbox[4]})
|
||||
|
||||
for bbox in equation_interline_bboxes:
|
||||
if any([bbox[0]>=bbox[2], bbox[1]>=bbox[3]]):
|
||||
logger.warning(f"equation_interline_bboxes: 错误的box, {bbox}")
|
||||
continue
|
||||
image_path = cut_image(bbox[:4], page_num, page, equation_interline_save_path, s3_return_equation_interline_path, img_s3_client, upload_switch=False)
|
||||
interline_eq_info.append({"bbox":bbox[:4], "image_path":image_path, "latex_text":bbox[4]})
|
||||
|
||||
return image_info, image_backup_info, table_info, inline_eq_info, interline_eq_info
|
||||
@@ -0,0 +1,11 @@
|
||||
import os
|
||||
|
||||
|
||||
def sanitize_filename(filename, replacement="_"):
|
||||
if os.name == 'nt':
|
||||
invalid_chars = '<>:"|?*'
|
||||
|
||||
for char in invalid_chars:
|
||||
filename = filename.replace(char, replacement)
|
||||
|
||||
return filename
|
||||
@@ -0,0 +1,33 @@
|
||||
import math
|
||||
|
||||
|
||||
def __inc_dict_val(mp, key, val_inc:int):
|
||||
if mp.get(key):
|
||||
mp[key] = mp[key] + val_inc
|
||||
else:
|
||||
mp[key] = val_inc
|
||||
|
||||
|
||||
|
||||
def get_text_block_base_info(block):
|
||||
"""
|
||||
获取这个文本块里的字体的颜色、字号、字体
|
||||
按照正文字数最多的返回
|
||||
"""
|
||||
|
||||
counter = {}
|
||||
|
||||
for line in block['lines']:
|
||||
for span in line['spans']:
|
||||
color = span['color']
|
||||
size = round(span['size'], 2)
|
||||
font = span['font']
|
||||
|
||||
txt_len = len(span['text'])
|
||||
__inc_dict_val(counter, (color, size, font), txt_len)
|
||||
|
||||
|
||||
c, s, ft = max(counter, key=counter.get)
|
||||
|
||||
return c, s, ft
|
||||
|
||||
@@ -0,0 +1,310 @@
|
||||
from libs.commons import fitz
|
||||
import os
|
||||
from loguru import logger
|
||||
from layout.bbox_sort import CONTENT_TYPE_IDX
|
||||
|
||||
|
||||
def draw_bbox_on_page(raw_pdf_doc: fitz.Document, paras_dict:dict, save_path: str):
|
||||
"""
|
||||
在page上画出bbox,保存到save_path
|
||||
"""
|
||||
# 检查文件是否存在
|
||||
is_new_pdf = False
|
||||
if os.path.exists(save_path):
|
||||
# 打开现有的 PDF 文件
|
||||
doc = fitz.open(save_path)
|
||||
else:
|
||||
# 创建一个新的空白 PDF 文件
|
||||
is_new_pdf = True
|
||||
doc = fitz.open('')
|
||||
|
||||
color_map = {
|
||||
'image': fitz.pdfcolor["yellow"],
|
||||
'text': fitz.pdfcolor['blue'],
|
||||
"table": fitz.pdfcolor['green']
|
||||
}
|
||||
|
||||
for k, v in paras_dict.items():
|
||||
page_idx = v['page_idx']
|
||||
width = raw_pdf_doc[page_idx].rect.width
|
||||
height = raw_pdf_doc[page_idx].rect.height
|
||||
new_page = doc.new_page(width=width, height=height)
|
||||
|
||||
shape = new_page.new_shape()
|
||||
for order, block in enumerate(v['preproc_blocks']):
|
||||
rect = fitz.Rect(block['bbox'])
|
||||
shape = new_page.new_shape()
|
||||
shape.draw_rect(rect)
|
||||
shape.finish(color=None, fill=color_map['text'], fill_opacity=0.2)
|
||||
shape.finish()
|
||||
shape.commit()
|
||||
|
||||
for img in v['images']:
|
||||
# 原始box画上去
|
||||
rect = fitz.Rect(img['bbox'])
|
||||
shape = new_page.new_shape()
|
||||
shape.draw_rect(rect)
|
||||
shape.finish(color=None, fill=fitz.pdfcolor['yellow'])
|
||||
shape.finish()
|
||||
shape.commit()
|
||||
|
||||
for img in v['image_backup']:
|
||||
# 原始box画上去
|
||||
rect = fitz.Rect(img['bbox'])
|
||||
shape = new_page.new_shape()
|
||||
shape.draw_rect(rect)
|
||||
shape.finish(color=fitz.pdfcolor['yellow'], fill=None)
|
||||
shape.finish()
|
||||
shape.commit()
|
||||
|
||||
for tb in v['droped_text_block']:
|
||||
# 原始box画上去
|
||||
rect = fitz.Rect(tb['bbox'])
|
||||
shape = new_page.new_shape()
|
||||
shape.draw_rect(rect)
|
||||
shape.finish(color=None, fill=fitz.pdfcolor['black'], fill_opacity=0.4)
|
||||
shape.finish()
|
||||
shape.commit()
|
||||
|
||||
# TODO table
|
||||
for tb in v['tables']:
|
||||
rect = fitz.Rect(tb['bbox'])
|
||||
shape = new_page.new_shape()
|
||||
shape.draw_rect(rect)
|
||||
shape.finish(color=None, fill=fitz.pdfcolor['green'], fill_opacity=0.2)
|
||||
shape.finish()
|
||||
shape.commit()
|
||||
|
||||
|
||||
parent_dir = os.path.dirname(save_path)
|
||||
if not os.path.exists(parent_dir):
|
||||
os.makedirs(parent_dir)
|
||||
|
||||
if is_new_pdf:
|
||||
doc.save(save_path)
|
||||
else:
|
||||
doc.saveIncr()
|
||||
doc.close()
|
||||
|
||||
|
||||
def debug_show_bbox(raw_pdf_doc: fitz.Document, page_idx: int, bboxes: list, droped_bboxes:list, expect_drop_bboxes:list, save_path: str, expected_page_id:int):
|
||||
"""
|
||||
以覆盖的方式写个临时的pdf,用于debug
|
||||
"""
|
||||
if page_idx!=expected_page_id:
|
||||
return
|
||||
|
||||
if os.path.exists(save_path):
|
||||
# 删除已经存在的文件
|
||||
os.remove(save_path)
|
||||
# 创建一个新的空白 PDF 文件
|
||||
doc = fitz.open('')
|
||||
|
||||
width = raw_pdf_doc[page_idx].rect.width
|
||||
height = raw_pdf_doc[page_idx].rect.height
|
||||
new_page = doc.new_page(width=width, height=height)
|
||||
|
||||
shape = new_page.new_shape()
|
||||
for bbox in bboxes:
|
||||
# 原始box画上去
|
||||
rect = fitz.Rect(*bbox[0:4])
|
||||
shape = new_page.new_shape()
|
||||
shape.draw_rect(rect)
|
||||
shape.finish(color=fitz.pdfcolor['red'], fill=fitz.pdfcolor['blue'], fill_opacity=0.2)
|
||||
shape.finish()
|
||||
shape.commit()
|
||||
|
||||
for bbox in droped_bboxes:
|
||||
# 原始box画上去
|
||||
rect = fitz.Rect(*bbox[0:4])
|
||||
shape = new_page.new_shape()
|
||||
shape.draw_rect(rect)
|
||||
shape.finish(color=None, fill=fitz.pdfcolor['yellow'], fill_opacity=0.2)
|
||||
shape.finish()
|
||||
shape.commit()
|
||||
|
||||
for bbox in expect_drop_bboxes:
|
||||
# 原始box画上去
|
||||
rect = fitz.Rect(*bbox[0:4])
|
||||
shape = new_page.new_shape()
|
||||
shape.draw_rect(rect)
|
||||
shape.finish(color=fitz.pdfcolor['red'], fill=None)
|
||||
shape.finish()
|
||||
shape.commit()
|
||||
|
||||
# shape.insert_textbox(fitz.Rect(200, 0, 600, 20), f"total bboxes: {len(bboxes)}", fontname="helv", fontsize=12,
|
||||
# color=(0, 0, 0))
|
||||
# shape.finish(color=fitz.pdfcolor['black'])
|
||||
# shape.commit()
|
||||
|
||||
parent_dir = os.path.dirname(save_path)
|
||||
if not os.path.exists(parent_dir):
|
||||
os.makedirs(parent_dir)
|
||||
|
||||
doc.save(save_path)
|
||||
doc.close()
|
||||
|
||||
|
||||
def debug_show_page(page, bboxes1: list,bboxes2: list,bboxes3: list,):
|
||||
save_path = "./tmp/debug.pdf"
|
||||
if os.path.exists(save_path):
|
||||
# 删除已经存在的文件
|
||||
os.remove(save_path)
|
||||
# 创建一个新的空白 PDF 文件
|
||||
doc = fitz.open('')
|
||||
|
||||
width = page.rect.width
|
||||
height = page.rect.height
|
||||
new_page = doc.new_page(width=width, height=height)
|
||||
|
||||
shape = new_page.new_shape()
|
||||
for bbox in bboxes1:
|
||||
# 原始box画上去
|
||||
rect = fitz.Rect(*bbox[0:4])
|
||||
shape = new_page.new_shape()
|
||||
shape.draw_rect(rect)
|
||||
shape.finish(color=fitz.pdfcolor['red'], fill=fitz.pdfcolor['blue'], fill_opacity=0.2)
|
||||
shape.finish()
|
||||
shape.commit()
|
||||
|
||||
for bbox in bboxes2:
|
||||
# 原始box画上去
|
||||
rect = fitz.Rect(*bbox[0:4])
|
||||
shape = new_page.new_shape()
|
||||
shape.draw_rect(rect)
|
||||
shape.finish(color=None, fill=fitz.pdfcolor['yellow'], fill_opacity=0.2)
|
||||
shape.finish()
|
||||
shape.commit()
|
||||
|
||||
for bbox in bboxes3:
|
||||
# 原始box画上去
|
||||
rect = fitz.Rect(*bbox[0:4])
|
||||
shape = new_page.new_shape()
|
||||
shape.draw_rect(rect)
|
||||
shape.finish(color=fitz.pdfcolor['red'], fill=None)
|
||||
shape.finish()
|
||||
shape.commit()
|
||||
|
||||
parent_dir = os.path.dirname(save_path)
|
||||
if not os.path.exists(parent_dir):
|
||||
os.makedirs(parent_dir)
|
||||
|
||||
doc.save(save_path)
|
||||
doc.close()
|
||||
|
||||
|
||||
|
||||
|
||||
def draw_layout_bbox_on_page(raw_pdf_doc: fitz.Document, paras_dict:dict, header, footer, pdf_path: str):
|
||||
"""
|
||||
在page上画出bbox,保存到save_path
|
||||
"""
|
||||
# 检查文件是否存在
|
||||
is_new_pdf = False
|
||||
if os.path.exists(pdf_path):
|
||||
# 打开现有的 PDF 文件
|
||||
doc = fitz.open(pdf_path)
|
||||
else:
|
||||
# 创建一个新的空白 PDF 文件
|
||||
is_new_pdf = True
|
||||
doc = fitz.open('')
|
||||
|
||||
for k, v in paras_dict.items():
|
||||
page_idx = v['page_idx']
|
||||
layouts = v['layout_bboxes']
|
||||
page = doc[page_idx]
|
||||
shape = page.new_shape()
|
||||
for order, layout in enumerate(layouts):
|
||||
border_offset = 1
|
||||
rect_box = layout['layout_bbox']
|
||||
layout_label = layout['layout_label']
|
||||
fill_color = fitz.pdfcolor['pink'] if layout_label=='U' else None
|
||||
rect_box = [rect_box[0]+1, rect_box[1]-border_offset, rect_box[2]-1, rect_box[3]+border_offset]
|
||||
rect = fitz.Rect(*rect_box)
|
||||
shape.draw_rect(rect)
|
||||
shape.finish(color=fitz.pdfcolor['red'], fill=fill_color, fill_opacity=0.4)
|
||||
"""
|
||||
draw order text on layout box
|
||||
"""
|
||||
font_size = 10
|
||||
shape.insert_text((rect_box[0] + 1, rect_box[1] + font_size), f"{order}", fontsize=font_size, color=(0, 0, 0))
|
||||
|
||||
"""画上footer header"""
|
||||
if header:
|
||||
shape.draw_rect(fitz.Rect(header))
|
||||
shape.finish(color=None, fill=fitz.pdfcolor['black'], fill_opacity=0.2)
|
||||
if footer:
|
||||
shape.draw_rect(fitz.Rect(footer))
|
||||
shape.finish(color=None, fill=fitz.pdfcolor['black'], fill_opacity=0.2)
|
||||
|
||||
shape.commit()
|
||||
|
||||
if is_new_pdf:
|
||||
doc.save(pdf_path)
|
||||
else:
|
||||
doc.saveIncr()
|
||||
doc.close()
|
||||
|
||||
|
||||
@DeprecationWarning
|
||||
def draw_layout_on_page(raw_pdf_doc: fitz.Document, page_idx: int, page_layout: list, pdf_path: str):
|
||||
"""
|
||||
把layout的box用红色边框花在pdf_path的page_idx上
|
||||
"""
|
||||
def draw(shape, layout, fill_color=fitz.pdfcolor['pink']):
|
||||
border_offset = 1
|
||||
rect_box = layout['layout_bbox']
|
||||
layout_label = layout['layout_label']
|
||||
sub_layout = layout['sub_layout']
|
||||
if len(sub_layout)==0:
|
||||
fill_color = fill_color if layout_label=='U' else None
|
||||
rect_box = [rect_box[0]+1, rect_box[1]-border_offset, rect_box[2]-1, rect_box[3]+border_offset]
|
||||
rect = fitz.Rect(*rect_box)
|
||||
shape.draw_rect(rect)
|
||||
shape.finish(color=fitz.pdfcolor['red'], fill=fill_color, fill_opacity=0.2)
|
||||
# if layout_label=='U':
|
||||
# bad_boxes = layout.get("bad_boxes", [])
|
||||
# for bad_box in bad_boxes:
|
||||
# rect = fitz.Rect(*bad_box)
|
||||
# shape.draw_rect(rect)
|
||||
# shape.finish(color=fitz.pdfcolor['red'], fill=fitz.pdfcolor['red'], fill_opacity=0.2)
|
||||
# else:
|
||||
# rect = fitz.Rect(*rect_box)
|
||||
# shape.draw_rect(rect)
|
||||
# shape.finish(color=fitz.pdfcolor['blue'])
|
||||
|
||||
for sub_layout in sub_layout:
|
||||
draw(shape, sub_layout)
|
||||
shape.commit()
|
||||
|
||||
|
||||
# 检查文件是否存在
|
||||
is_new_pdf = False
|
||||
if os.path.exists(pdf_path):
|
||||
# 打开现有的 PDF 文件
|
||||
doc = fitz.open(pdf_path)
|
||||
else:
|
||||
# 创建一个新的空白 PDF 文件
|
||||
is_new_pdf = True
|
||||
doc = fitz.open('')
|
||||
|
||||
page = doc[page_idx]
|
||||
shape = page.new_shape()
|
||||
for order, layout in enumerate(page_layout):
|
||||
draw(shape, layout, fitz.pdfcolor['yellow'])
|
||||
|
||||
# shape.insert_textbox(fitz.Rect(200, 0, 600, 20), f"total bboxes: {len(layout)}", fontname="helv", fontsize=12,
|
||||
# color=(0, 0, 0))
|
||||
# shape.finish(color=fitz.pdfcolor['black'])
|
||||
# shape.commit()
|
||||
|
||||
parent_dir = os.path.dirname(pdf_path)
|
||||
if not os.path.exists(parent_dir):
|
||||
os.makedirs(parent_dir)
|
||||
|
||||
if is_new_pdf:
|
||||
doc.save(pdf_path)
|
||||
else:
|
||||
doc.saveIncr()
|
||||
doc.close()
|
||||
|
||||
+250
@@ -0,0 +1,250 @@
|
||||
import re
|
||||
import math
|
||||
from loguru import logger
|
||||
|
||||
from libs.boxbase import find_bottom_nearest_text_bbox, find_top_nearest_text_bbox
|
||||
|
||||
|
||||
def mk_nlp_markdown(para_dict: dict):
|
||||
"""
|
||||
对排序后的bboxes拼接内容
|
||||
"""
|
||||
content_lst = []
|
||||
for _, page_info in para_dict.items():
|
||||
para_blocks = page_info.get("para_blocks")
|
||||
if not para_blocks:
|
||||
continue
|
||||
|
||||
for block in para_blocks:
|
||||
item = block["paras"]
|
||||
for _, p in item.items():
|
||||
para_text = p["para_text"]
|
||||
is_title = p["is_para_title"]
|
||||
title_level = p['para_title_level']
|
||||
md_title_prefix = "#"*title_level
|
||||
if is_title:
|
||||
content_lst.append(f"{md_title_prefix} {para_text}")
|
||||
else:
|
||||
content_lst.append(para_text)
|
||||
|
||||
content_text = "\n\n".join(content_lst)
|
||||
|
||||
return content_text
|
||||
|
||||
|
||||
|
||||
# 找到目标字符串在段落中的索引
|
||||
def __find_index(paragraph, target):
|
||||
index = paragraph.find(target)
|
||||
if index != -1:
|
||||
return index
|
||||
else:
|
||||
return None
|
||||
|
||||
|
||||
def __insert_string(paragraph, target, postion):
|
||||
new_paragraph = paragraph[:postion] + target + paragraph[postion:]
|
||||
return new_paragraph
|
||||
|
||||
|
||||
def __insert_after(content, image_content, target):
|
||||
"""
|
||||
在content中找到target,将image_content插入到target后面
|
||||
"""
|
||||
index = content.find(target)
|
||||
if index != -1:
|
||||
content = content[:index+len(target)] + "\n\n" + image_content + "\n\n" + content[index+len(target):]
|
||||
else:
|
||||
logger.error(f"Can't find the location of image {image_content} in the markdown file, search target is {target}")
|
||||
return content
|
||||
|
||||
def __insert_before(content, image_content, target):
|
||||
"""
|
||||
在content中找到target,将image_content插入到target前面
|
||||
"""
|
||||
index = content.find(target)
|
||||
if index != -1:
|
||||
content = content[:index] + "\n\n" + image_content + "\n\n" + content[index:]
|
||||
else:
|
||||
logger.error(f"Can't find the location of image {image_content} in the markdown file, search target is {target}")
|
||||
return content
|
||||
|
||||
|
||||
|
||||
def mk_mm_markdown(para_dict: dict):
|
||||
"""拼装多模态markdown"""
|
||||
content_lst = []
|
||||
for _, page_info in para_dict.items():
|
||||
page_lst = [] # 一个page内的段落列表
|
||||
para_blocks = page_info.get("para_blocks")
|
||||
pymu_raw_blocks = page_info.get("preproc_blocks")
|
||||
|
||||
all_page_images = []
|
||||
all_page_images.extend(page_info.get("images",[]))
|
||||
all_page_images.extend(page_info.get("image_backup", []) )
|
||||
all_page_images.extend(page_info.get("tables",[]))
|
||||
all_page_images.extend(page_info.get("table_backup",[]) )
|
||||
|
||||
if not para_blocks or not pymu_raw_blocks: # 只有图片的拼接的场景
|
||||
for img in all_page_images:
|
||||
page_lst.append(f"") # TODO 图片顺序
|
||||
page_md = "\n\n".join(page_lst)
|
||||
|
||||
else:
|
||||
for block in para_blocks:
|
||||
item = block["paras"]
|
||||
for _, p in item.items():
|
||||
para_text = p["para_text"]
|
||||
is_title = p["is_para_title"]
|
||||
title_level = p['para_title_level']
|
||||
md_title_prefix = "#"*title_level
|
||||
if is_title:
|
||||
page_lst.append(f"{md_title_prefix} {para_text}")
|
||||
else:
|
||||
page_lst.append(para_text)
|
||||
|
||||
"""拼装成一个页面的文本"""
|
||||
page_md = "\n\n".join(page_lst)
|
||||
"""插入图片"""
|
||||
for img in all_page_images:
|
||||
imgbox = img['bbox']
|
||||
img_content = f""
|
||||
# 先看在哪个block内
|
||||
for block in pymu_raw_blocks:
|
||||
bbox = block['bbox']
|
||||
if bbox[0]-1 <= imgbox[0] < bbox[2]+1 and bbox[1]-1 <= imgbox[1] < bbox[3]+1:# 确定在block内
|
||||
for l in block['lines']:
|
||||
line_box = l['bbox']
|
||||
if line_box[0]-1 <= imgbox[0] < line_box[2]+1 and line_box[1]-1 <= imgbox[1] < line_box[3]+1: # 在line内的,插入line前面
|
||||
line_txt = "".join([s['text'] for s in l['spans']])
|
||||
page_md = __insert_before(page_md, img_content, line_txt)
|
||||
break
|
||||
break
|
||||
else:# 在行与行之间
|
||||
# 找到图片x0,y0与line的x0,y0最近的line
|
||||
min_distance = 100000
|
||||
min_line = None
|
||||
for l in block['lines']:
|
||||
line_box = l['bbox']
|
||||
distance = math.sqrt((line_box[0] - imgbox[0])**2 + (line_box[1] - imgbox[1])**2)
|
||||
if distance < min_distance:
|
||||
min_distance = distance
|
||||
min_line = l
|
||||
if min_line:
|
||||
line_txt = "".join([s['text'] for s in min_line['spans']])
|
||||
img_h = imgbox[3] - imgbox[1]
|
||||
if min_distance<img_h: # 文字在图片前面
|
||||
page_md = __insert_after(page_md, img_content, line_txt)
|
||||
else:
|
||||
page_md = __insert_before(page_md, img_content, line_txt)
|
||||
else:
|
||||
logger.error(f"Can't find the location of image {img['image_path']} in the markdown file")
|
||||
else:# 应当在两个block之间
|
||||
# 找到上方最近的block,如果上方没有就找大下方最近的block
|
||||
top_txt_block = find_top_nearest_text_bbox(pymu_raw_blocks, imgbox)
|
||||
if top_txt_block:
|
||||
line_txt = "".join([s['text'] for s in top_txt_block['lines'][-1]['spans']])
|
||||
page_md = __insert_after(page_md, img_content, line_txt)
|
||||
else:
|
||||
bottom_txt_block = find_bottom_nearest_text_bbox(pymu_raw_blocks, imgbox)
|
||||
if bottom_txt_block:
|
||||
line_txt = "".join([s['text'] for s in bottom_txt_block['lines'][0]['spans']])
|
||||
page_md = __insert_before(page_md, img_content, line_txt)
|
||||
else:
|
||||
logger.error(f"Can't find the location of image {img['image_path']} in the markdown file")
|
||||
|
||||
content_lst.append(page_md)
|
||||
|
||||
"""拼装成全部页面的文本"""
|
||||
content_text = "\n\n".join(content_lst)
|
||||
|
||||
return content_text
|
||||
|
||||
|
||||
@DeprecationWarning
|
||||
def mk_mm_markdown_1(para_dict: dict):
|
||||
"""
|
||||
得到images和tables变量
|
||||
"""
|
||||
image_all_list = []
|
||||
|
||||
for _, page_info in para_dict.items():
|
||||
images = page_info.get("images",[])
|
||||
tables = page_info.get("tables",[])
|
||||
image_backup = page_info.get("image_backup", [])
|
||||
table_backup = page_info.get("table_backup",[])
|
||||
all_page_images = []
|
||||
all_page_images.extend(images)
|
||||
all_page_images.extend(image_backup)
|
||||
all_page_images.extend(tables)
|
||||
all_page_images.extend(table_backup)
|
||||
|
||||
pymu_raw_blocks = page_info.get("pymu_raw_blocks")
|
||||
|
||||
# 提取每个图片所在位置
|
||||
for image_info in all_page_images:
|
||||
x0_image, y0_image, x1_image, y1_image = image_info['bbox'][:4]
|
||||
image_path = image_info['image_path']
|
||||
|
||||
# 判断图片处于原始PDF中哪个模块之间
|
||||
image_internal_dict = {}
|
||||
image_external_dict = {}
|
||||
between_dict = {}
|
||||
for block in pymu_raw_blocks:
|
||||
x0, y0, x1, y1 = block['bbox'][:4]
|
||||
|
||||
# 在某个模块内部
|
||||
if x0 <= x0_image < x1 and y0 <= y0_image < y1:
|
||||
image_internal_dict['bbox'] = [x0_image, y0_image, x1_image, y1_image]
|
||||
image_internal_dict['path'] = image_path
|
||||
|
||||
# 确定图片在哪句文本之前
|
||||
y_pre = 0
|
||||
for line in block['lines']:
|
||||
x0, y0, x1, y1 = line['spans'][0]['bbox']
|
||||
if x0 <= x0_image < x1 and y_pre <= y0_image < y0:
|
||||
text = line['spans']['text']
|
||||
image_internal_dict['text'] = text
|
||||
image_internal_dict['markdown_image'] = f''
|
||||
break
|
||||
else:
|
||||
y_pre = y0
|
||||
# 在某两个模块之间
|
||||
elif x0 <= x0_image < x1:
|
||||
distance = math.sqrt((x1_image - x0)**2 + (y1_image - y0)**2)
|
||||
between_dict[block['number']] = distance
|
||||
|
||||
# 找到与定位点距离最小的文本block
|
||||
if between_dict:
|
||||
min_key = min(between_dict, key=between_dict.get)
|
||||
spans_list = []
|
||||
for span in pymu_raw_blocks[min_key]['lines']:
|
||||
for text_piece in span['spans']:
|
||||
# 防止索引定位文本内容过多
|
||||
if len(spans_list) < 60:
|
||||
spans_list.append(text_piece['text'])
|
||||
text1 = ''.join(spans_list)
|
||||
|
||||
image_external_dict['bbox'] = [x0_image, y0_image, x1_image, y1_image]
|
||||
image_external_dict['path'] = image_path
|
||||
image_external_dict['text'] = text1
|
||||
image_external_dict['markdown_image'] = f''
|
||||
|
||||
# 将内部图片或外部图片存入当页所有图片的列表
|
||||
if len(image_internal_dict) != 0:
|
||||
image_all_list.append(image_internal_dict)
|
||||
elif len(image_external_dict) != 0:
|
||||
image_all_list.append(image_external_dict)
|
||||
else:
|
||||
logger.error(f"Can't find the location of image {image_path} in the markdown file")
|
||||
|
||||
content_text = mk_nlp_markdown(para_dict)
|
||||
|
||||
for image_info_extract in image_all_list:
|
||||
loc = __find_index(content_text, image_info_extract['text'])
|
||||
if loc is not None:
|
||||
content_text = __insert_string(content_text, image_info_extract['markdown_image'], loc)
|
||||
else:
|
||||
logger.error(f"Can't find the location of image {image_info_extract['path']} in the markdown file")
|
||||
|
||||
return content_text
|
||||
@@ -0,0 +1,563 @@
|
||||
import os
|
||||
import sys
|
||||
import unicodedata
|
||||
|
||||
from para.commons import *
|
||||
|
||||
|
||||
if sys.version_info[0] >= 3:
|
||||
sys.stdout.reconfigure(encoding="utf-8") # type: ignore
|
||||
|
||||
|
||||
class BlockContinuationProcessor:
|
||||
"""
|
||||
This class is used to process the blocks to detect block continuations.
|
||||
"""
|
||||
|
||||
def __init__(self) -> None:
|
||||
pass
|
||||
|
||||
def __is_similar_font_type(self, font_type1, font_type2, prefix_length_ratio=0.3):
|
||||
"""
|
||||
This function checks if the two font types are similar.
|
||||
Definition of similar font types: the two font types have a common prefix,
|
||||
and the length of the common prefix is at least a certain ratio of the length of the shorter font type.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
font_type1 : str
|
||||
font type 1
|
||||
font_type2 : str
|
||||
font type 2
|
||||
prefix_length_ratio : float
|
||||
minimum ratio of the common prefix length to the length of the shorter font type
|
||||
|
||||
Returns
|
||||
-------
|
||||
bool
|
||||
True if the two font types are similar, False otherwise.
|
||||
"""
|
||||
|
||||
if isinstance(font_type1, list):
|
||||
font_type1 = font_type1[0] if font_type1 else ""
|
||||
if isinstance(font_type2, list):
|
||||
font_type2 = font_type2[0] if font_type2 else ""
|
||||
|
||||
if font_type1 == font_type2:
|
||||
return True
|
||||
|
||||
# Find the length of the common prefix
|
||||
common_prefix_length = len(os.path.commonprefix([font_type1, font_type2]))
|
||||
|
||||
# Calculate the minimum prefix length based on the ratio
|
||||
min_prefix_length = int(min(len(font_type1), len(font_type2)) * prefix_length_ratio)
|
||||
|
||||
return common_prefix_length >= min_prefix_length
|
||||
|
||||
def __is_same_block_font(self, block1, block2):
|
||||
"""
|
||||
This function compares the font of block1 and block2
|
||||
|
||||
Parameters
|
||||
----------
|
||||
block1 : dict
|
||||
block1
|
||||
block2 : dict
|
||||
block2
|
||||
|
||||
Returns
|
||||
-------
|
||||
is_same : bool
|
||||
True if block1 and block2 have the same font, else False
|
||||
"""
|
||||
block_1_font_type = safe_get(block1, "block_font_type", "")
|
||||
block_1_font_size = safe_get(block1, "block_font_size", 0)
|
||||
block_1_avg_char_width = safe_get(block1, "avg_char_width", 0)
|
||||
|
||||
block_2_font_type = safe_get(block2, "block_font_type", "")
|
||||
block_2_font_size = safe_get(block2, "block_font_size", 0)
|
||||
block_2_avg_char_width = safe_get(block2, "avg_char_width", 0)
|
||||
|
||||
if isinstance(block_1_font_size, list):
|
||||
block_1_font_size = block_1_font_size[0] if block_1_font_size else 0
|
||||
if isinstance(block_2_font_size, list):
|
||||
block_2_font_size = block_2_font_size[0] if block_2_font_size else 0
|
||||
|
||||
block_1_text = safe_get(block1, "text", "")
|
||||
block_2_text = safe_get(block2, "text", "")
|
||||
|
||||
if block_1_avg_char_width == 0 or block_2_avg_char_width == 0:
|
||||
return False
|
||||
|
||||
if not block_1_text or not block_2_text:
|
||||
return False
|
||||
else:
|
||||
text_len_ratio = len(block_2_text) / len(block_1_text)
|
||||
if text_len_ratio < 0.2:
|
||||
avg_char_width_condition = (
|
||||
abs(block_1_avg_char_width - block_2_avg_char_width) / min(block_1_avg_char_width, block_2_avg_char_width)
|
||||
< 0.5
|
||||
)
|
||||
else:
|
||||
avg_char_width_condition = (
|
||||
abs(block_1_avg_char_width - block_2_avg_char_width) / min(block_1_avg_char_width, block_2_avg_char_width)
|
||||
< 0.2
|
||||
)
|
||||
|
||||
block_font_size_condtion = abs(block_1_font_size - block_2_font_size) < 1
|
||||
|
||||
return (
|
||||
self.__is_similar_font_type(block_1_font_type, block_2_font_type)
|
||||
and avg_char_width_condition
|
||||
and block_font_size_condtion
|
||||
)
|
||||
|
||||
def _is_alphabet_char(self, char):
|
||||
if (char >= "\u0041" and char <= "\u005a") or (char >= "\u0061" and char <= "\u007a"):
|
||||
return True
|
||||
else:
|
||||
return False
|
||||
|
||||
def _is_chinese_char(self, char):
|
||||
if char >= "\u4e00" and char <= "\u9fa5":
|
||||
return True
|
||||
else:
|
||||
return False
|
||||
|
||||
def _is_other_letter_char(self, char):
|
||||
try:
|
||||
cat = unicodedata.category(char)
|
||||
if cat == "Lu" or cat == "Ll":
|
||||
return not self._is_alphabet_char(char) and not self._is_chinese_char(char)
|
||||
except TypeError:
|
||||
print("The input to the function must be a single character.")
|
||||
return False
|
||||
|
||||
def _is_year(self, s: str):
|
||||
try:
|
||||
number = int(s)
|
||||
return 1900 <= number <= 2099
|
||||
except ValueError:
|
||||
return False
|
||||
|
||||
def __is_para_font_consistent(self, para_1, para_2):
|
||||
"""
|
||||
This function compares the font of para1 and para2
|
||||
|
||||
Parameters
|
||||
----------
|
||||
para1 : dict
|
||||
para1
|
||||
para2 : dict
|
||||
para2
|
||||
|
||||
Returns
|
||||
-------
|
||||
is_same : bool
|
||||
True if para1 and para2 have the same font, else False
|
||||
"""
|
||||
if para_1 is None or para_2 is None:
|
||||
return False
|
||||
|
||||
para_1_font_type = safe_get(para_1, "para_font_type", "")
|
||||
para_1_font_size = safe_get(para_1, "para_font_size", 0)
|
||||
para_1_font_color = safe_get(para_1, "para_font_color", "")
|
||||
|
||||
para_2_font_type = safe_get(para_2, "para_font_type", "")
|
||||
para_2_font_size = safe_get(para_2, "para_font_size", 0)
|
||||
para_2_font_color = safe_get(para_2, "para_font_color", "")
|
||||
|
||||
if isinstance(para_1_font_type, list): # get the most common font type
|
||||
para_1_font_type = max(set(para_1_font_type), key=para_1_font_type.count)
|
||||
if isinstance(para_2_font_type, list):
|
||||
para_2_font_type = max(set(para_2_font_type), key=para_2_font_type.count)
|
||||
if isinstance(para_1_font_size, list): # compute average font type
|
||||
para_1_font_size = sum(para_1_font_size) / len(para_1_font_size)
|
||||
if isinstance(para_2_font_size, list): # compute average font type
|
||||
para_2_font_size = sum(para_2_font_size) / len(para_2_font_size)
|
||||
|
||||
return (
|
||||
self.__is_similar_font_type(para_1_font_type, para_2_font_type)
|
||||
and abs(para_1_font_size - para_2_font_size) < 1.5
|
||||
# and para_font_color1 == para_font_color2
|
||||
)
|
||||
|
||||
def _is_para_puncs_consistent(self, para_1, para_2):
|
||||
"""
|
||||
This function determines whether para1 and para2 are originally from the same paragraph by checking the puncs of para1(former) and para2(latter)
|
||||
|
||||
Parameters
|
||||
----------
|
||||
para1 : dict
|
||||
para1
|
||||
para2 : dict
|
||||
para2
|
||||
|
||||
Returns
|
||||
-------
|
||||
is_same : bool
|
||||
True if para1 and para2 are from the same paragraph by using the puncs, else False
|
||||
"""
|
||||
para_1_text = safe_get(para_1, "para_text", "").strip()
|
||||
para_2_text = safe_get(para_2, "para_text", "").strip()
|
||||
|
||||
para_1_bboxes = safe_get(para_1, "para_bbox", [])
|
||||
para_1_font_sizes = safe_get(para_1, "para_font_size", 0)
|
||||
|
||||
para_2_bboxes = safe_get(para_2, "para_bbox", [])
|
||||
para_2_font_sizes = safe_get(para_2, "para_font_size", 0)
|
||||
|
||||
# print_yellow(" Features of determine puncs_consistent:")
|
||||
# print(f" para_1_text: {para_1_text}")
|
||||
# print(f" para_2_text: {para_2_text}")
|
||||
# print(f" para_1_bboxes: {para_1_bboxes}")
|
||||
# print(f" para_2_bboxes: {para_2_bboxes}")
|
||||
# print(f" para_1_font_sizes: {para_1_font_sizes}")
|
||||
# print(f" para_2_font_sizes: {para_2_font_sizes}")
|
||||
|
||||
if is_nested_list(para_1_bboxes):
|
||||
x0_1, y0_1, x1_1, y1_1 = para_1_bboxes[-1]
|
||||
else:
|
||||
x0_1, y0_1, x1_1, y1_1 = para_1_bboxes
|
||||
|
||||
if is_nested_list(para_2_bboxes):
|
||||
x0_2, y0_2, x1_2, y1_2 = para_2_bboxes[0]
|
||||
para_2_font_sizes = para_2_font_sizes[0] # type: ignore
|
||||
else:
|
||||
x0_2, y0_2, x1_2, y1_2 = para_2_bboxes
|
||||
|
||||
right_align_threshold = 0.5 * (para_1_font_sizes + para_2_font_sizes) * 0.8
|
||||
are_two_paras_right_aligned = abs(x1_1 - x1_2) < right_align_threshold
|
||||
|
||||
left_indent_threshold = 0.5 * (para_1_font_sizes + para_2_font_sizes) * 0.8
|
||||
is_para1_left_indent_than_papa2 = x0_1 - x0_2 > left_indent_threshold
|
||||
is_para2_left_indent_than_papa1 = x0_2 - x0_1 > left_indent_threshold
|
||||
|
||||
# Check if either para_text1 or para_text2 is empty
|
||||
if not para_1_text or not para_2_text:
|
||||
return False
|
||||
|
||||
# Define the end puncs for a sentence to end and hyphen
|
||||
end_puncs = [".", "?", "!", "。", "?", "!", "…"]
|
||||
hyphen = ["-", "—"]
|
||||
|
||||
# Check if para_text1 ends with either hyphen or non-end punctuation or spaces
|
||||
para_1_end_with_hyphen = para_1_text and para_1_text[-1] in hyphen
|
||||
para_1_end_with_end_punc = para_1_text and para_1_text[-1] in end_puncs
|
||||
para_1_end_with_space = para_1_text and para_1_text[-1] == " "
|
||||
para_1_not_end_with_end_punc = para_1_text and para_1_text[-1] not in end_puncs
|
||||
|
||||
# print_yellow(f" para_1_end_with_hyphen: {para_1_end_with_hyphen}")
|
||||
# print_yellow(f" para_1_end_with_end_punc: {para_1_end_with_end_punc}")
|
||||
# print_yellow(f" para_1_not_end_with_end_punc: {para_1_not_end_with_end_punc}")
|
||||
# print_yellow(f" para_1_end_with_space: {para_1_end_with_space}")
|
||||
|
||||
if para_1_end_with_hyphen: # If para_text1 ends with hyphen
|
||||
# print_red(f"para_1 is end with hyphen.")
|
||||
para_2_is_consistent = para_2_text and (
|
||||
para_2_text[0] in hyphen
|
||||
or (self._is_alphabet_char(para_2_text[0]) and para_2_text[0].islower())
|
||||
or (self._is_chinese_char(para_2_text[0]))
|
||||
or (self._is_other_letter_char(para_2_text[0]))
|
||||
)
|
||||
if para_2_is_consistent:
|
||||
# print(f"para_2 is consistent.\n")
|
||||
return True
|
||||
else:
|
||||
# print(f"para_2 is not consistent.\n")
|
||||
pass
|
||||
|
||||
elif para_1_end_with_end_punc: # If para_text1 ends with ending punctuations
|
||||
# print_red(f"para_1 is end with end_punc.")
|
||||
para_2_is_consistent = (
|
||||
para_2_text
|
||||
and (
|
||||
para_2_text[0] == " "
|
||||
or (self._is_alphabet_char(para_2_text[0]) and para_2_text[0].isupper())
|
||||
or (self._is_chinese_char(para_2_text[0]))
|
||||
or (self._is_other_letter_char(para_2_text[0]))
|
||||
)
|
||||
and not is_para2_left_indent_than_papa1
|
||||
)
|
||||
if para_2_is_consistent:
|
||||
# print(f"para_2 is consistent.\n")
|
||||
return True
|
||||
else:
|
||||
# print(f"para_2 is not consistent.\n")
|
||||
pass
|
||||
|
||||
elif para_1_not_end_with_end_punc: # If para_text1 is not end with ending punctuations
|
||||
# print_red(f"para_1 is NOT end with end_punc.")
|
||||
para_2_is_consistent = para_2_text and (
|
||||
para_2_text[0] == " "
|
||||
or (self._is_alphabet_char(para_2_text[0]) and para_2_text[0].islower())
|
||||
or (self._is_alphabet_char(para_2_text[0]))
|
||||
or (self._is_year(para_2_text[0:4]))
|
||||
or (are_two_paras_right_aligned or is_para1_left_indent_than_papa2)
|
||||
or (self._is_chinese_char(para_2_text[0]))
|
||||
or (self._is_other_letter_char(para_2_text[0]))
|
||||
)
|
||||
if para_2_is_consistent:
|
||||
# print(f"para_2 is consistent.\n")
|
||||
return True
|
||||
else:
|
||||
# print(f"para_2 is not consistent.\n")
|
||||
pass
|
||||
|
||||
elif para_1_end_with_space: # If para_text1 ends with space
|
||||
# print_red(f"para_1 is end with space.")
|
||||
para_2_is_consistent = para_2_text and (
|
||||
para_2_text[0] == " "
|
||||
or (self._is_alphabet_char(para_2_text[0]) and para_2_text[0].islower())
|
||||
or (self._is_chinese_char(para_2_text[0]))
|
||||
or (self._is_other_letter_char(para_2_text[0]))
|
||||
)
|
||||
if para_2_is_consistent:
|
||||
# print(f"para_2 is consistent.\n")
|
||||
return True
|
||||
else:
|
||||
pass
|
||||
# print(f"para_2 is not consistent.\n")
|
||||
|
||||
return False
|
||||
|
||||
def _is_block_consistent(self, block1, block2):
|
||||
"""
|
||||
This function determines whether block1 and block2 are originally from the same block
|
||||
|
||||
Parameters
|
||||
----------
|
||||
block1 : dict
|
||||
block1s
|
||||
block2 : dict
|
||||
block2
|
||||
|
||||
Returns
|
||||
-------
|
||||
is_same : bool
|
||||
True if block1 and block2 are from the same block, else False
|
||||
"""
|
||||
return self.__is_same_block_font(block1, block2)
|
||||
|
||||
def _is_para_continued(self, para1, para2):
|
||||
"""
|
||||
This function determines whether para1 and para2 are originally from the same paragraph
|
||||
|
||||
Parameters
|
||||
----------
|
||||
para1 : dict
|
||||
para1
|
||||
para2 : dict
|
||||
para2
|
||||
|
||||
Returns
|
||||
-------
|
||||
is_same : bool
|
||||
True if para1 and para2 are from the same paragraph, else False
|
||||
"""
|
||||
is_para_font_consistent = self.__is_para_font_consistent(para1, para2)
|
||||
is_para_puncs_consistent = self._is_para_puncs_consistent(para1, para2)
|
||||
|
||||
return is_para_font_consistent and is_para_puncs_consistent
|
||||
|
||||
def _are_boundaries_of_block_consistent(self, block1, block2):
|
||||
"""
|
||||
This function checks if the boundaries of block1 and block2 are consistent
|
||||
|
||||
Parameters
|
||||
----------
|
||||
block1 : dict
|
||||
block1
|
||||
|
||||
block2 : dict
|
||||
block2
|
||||
|
||||
Returns
|
||||
-------
|
||||
is_consistent : bool
|
||||
True if the boundaries of block1 and block2 are consistent, else False
|
||||
"""
|
||||
|
||||
last_line_of_block1 = block1["lines"][-1]
|
||||
first_line_of_block2 = block2["lines"][0]
|
||||
|
||||
spans_of_last_line_of_block1 = last_line_of_block1["spans"]
|
||||
spans_of_first_line_of_block2 = first_line_of_block2["spans"]
|
||||
|
||||
font_type_of_last_line_of_block1 = spans_of_last_line_of_block1[0]["font"].lower()
|
||||
font_size_of_last_line_of_block1 = spans_of_last_line_of_block1[0]["size"]
|
||||
font_color_of_last_line_of_block1 = spans_of_last_line_of_block1[0]["color"]
|
||||
font_flags_of_last_line_of_block1 = spans_of_last_line_of_block1[0]["flags"]
|
||||
|
||||
font_type_of_first_line_of_block2 = spans_of_first_line_of_block2[0]["font"].lower()
|
||||
font_size_of_first_line_of_block2 = spans_of_first_line_of_block2[0]["size"]
|
||||
font_color_of_first_line_of_block2 = spans_of_first_line_of_block2[0]["color"]
|
||||
font_flags_of_first_line_of_block2 = spans_of_first_line_of_block2[0]["flags"]
|
||||
|
||||
return (
|
||||
self.__is_similar_font_type(font_type_of_last_line_of_block1, font_type_of_first_line_of_block2)
|
||||
and abs(font_size_of_last_line_of_block1 - font_size_of_first_line_of_block2) < 1
|
||||
# and font_color_of_last_line_of_block1 == font_color_of_first_line_of_block2
|
||||
and font_flags_of_last_line_of_block1 == font_flags_of_first_line_of_block2
|
||||
)
|
||||
|
||||
def _get_last_paragraph(self, block):
|
||||
"""
|
||||
Retrieves the last paragraph from a block.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
block : dict
|
||||
The block from which to retrieve the paragraph.
|
||||
|
||||
Returns
|
||||
-------
|
||||
dict
|
||||
The last paragraph of the block.
|
||||
"""
|
||||
if block["paras"]:
|
||||
last_para_key = list(block["paras"].keys())[-1]
|
||||
return block["paras"][last_para_key]
|
||||
else:
|
||||
return None
|
||||
|
||||
def _get_first_paragraph(self, block):
|
||||
"""
|
||||
Retrieves the first paragraph from a block.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
block : dict
|
||||
The block from which to retrieve the paragraph.
|
||||
|
||||
Returns
|
||||
-------
|
||||
dict
|
||||
The first paragraph of the block.
|
||||
"""
|
||||
if block["paras"]:
|
||||
first_para_key = list(block["paras"].keys())[0]
|
||||
return block["paras"][first_para_key]
|
||||
else:
|
||||
return None
|
||||
|
||||
def should_merge_next_para(self, curr_para, next_para):
|
||||
if self._is_para_continued(curr_para, next_para):
|
||||
return True
|
||||
else:
|
||||
return False
|
||||
|
||||
def batch_tag_paras(self, pdf_dict):
|
||||
the_last_page_id = len(pdf_dict) - 1
|
||||
|
||||
for curr_page_idx, (curr_page_id, curr_page_content) in enumerate(pdf_dict.items()):
|
||||
if curr_page_id.startswith("page_") and curr_page_content.get("para_blocks", []):
|
||||
para_blocks_of_curr_page = curr_page_content["para_blocks"]
|
||||
next_page_idx = curr_page_idx + 1
|
||||
next_page_id = f"page_{next_page_idx}"
|
||||
next_page_content = pdf_dict.get(next_page_id, {})
|
||||
|
||||
for i, current_block in enumerate(para_blocks_of_curr_page):
|
||||
for para_id, curr_para in current_block["paras"].items():
|
||||
curr_para["curr_para_location"] = [
|
||||
curr_page_idx,
|
||||
current_block["block_id"],
|
||||
int(para_id.split("_")[-1]),
|
||||
]
|
||||
curr_para["next_para_location"] = None # 默认设置为None
|
||||
curr_para["merge_next_para"] = False # 默认设置为False
|
||||
|
||||
next_block = para_blocks_of_curr_page[i + 1] if i < len(para_blocks_of_curr_page) - 1 else None
|
||||
|
||||
if next_block:
|
||||
curr_block_last_para_key = list(current_block["paras"].keys())[-1]
|
||||
curr_blk_last_para = current_block["paras"][curr_block_last_para_key]
|
||||
|
||||
next_block_first_para_key = list(next_block["paras"].keys())[0]
|
||||
next_blk_first_para = next_block["paras"][next_block_first_para_key]
|
||||
|
||||
if self.should_merge_next_para(curr_blk_last_para, next_blk_first_para):
|
||||
curr_blk_last_para["next_para_location"] = [
|
||||
curr_page_idx,
|
||||
next_block["block_id"],
|
||||
int(next_block_first_para_key.split("_")[-1]),
|
||||
]
|
||||
curr_blk_last_para["merge_next_para"] = True
|
||||
else:
|
||||
# Handle the case where the next block is in a different page
|
||||
curr_block_last_para_key = list(current_block["paras"].keys())[-1]
|
||||
curr_blk_last_para = current_block["paras"][curr_block_last_para_key]
|
||||
|
||||
while not next_page_content.get("para_blocks", []) and next_page_idx <= the_last_page_id:
|
||||
next_page_idx += 1
|
||||
next_page_id = f"page_{next_page_idx}"
|
||||
next_page_content = pdf_dict.get(next_page_id, {})
|
||||
|
||||
if next_page_content.get("para_blocks", []):
|
||||
next_blk_first_para_key = list(next_page_content["para_blocks"][0]["paras"].keys())[0]
|
||||
next_blk_first_para = next_page_content["para_blocks"][0]["paras"][next_blk_first_para_key]
|
||||
|
||||
if self.should_merge_next_para(curr_blk_last_para, next_blk_first_para):
|
||||
curr_blk_last_para["next_para_location"] = [
|
||||
next_page_idx,
|
||||
next_page_content["para_blocks"][0]["block_id"],
|
||||
int(next_blk_first_para_key.split("_")[-1]),
|
||||
]
|
||||
curr_blk_last_para["merge_next_para"] = True
|
||||
|
||||
return pdf_dict
|
||||
|
||||
def find_block_by_id(self, para_blocks, block_id):
|
||||
for block in para_blocks:
|
||||
if block.get("block_id") == block_id:
|
||||
return block
|
||||
return None
|
||||
|
||||
def batch_merge_paras(self, pdf_dict):
|
||||
for page_id, page_content in pdf_dict.items():
|
||||
if page_id.startswith("page_") and page_content.get("para_blocks", []):
|
||||
para_blocks_of_page = page_content["para_blocks"]
|
||||
|
||||
for i in range(len(para_blocks_of_page)):
|
||||
current_block = para_blocks_of_page[i]
|
||||
paras = current_block["paras"]
|
||||
|
||||
for para_id, curr_para in list(paras.items()):
|
||||
# 跳过标题段落
|
||||
if curr_para.get("is_para_title"):
|
||||
continue
|
||||
|
||||
while curr_para.get("merge_next_para"):
|
||||
next_para_location = curr_para.get("next_para_location")
|
||||
if not next_para_location:
|
||||
break
|
||||
|
||||
next_page_idx, next_block_id, next_para_id = next_para_location
|
||||
next_page_id = f"page_{next_page_idx}"
|
||||
next_page_content = pdf_dict.get(next_page_id)
|
||||
if not next_page_content:
|
||||
break
|
||||
|
||||
next_block = self.find_block_by_id(next_page_content.get("para_blocks", []), next_block_id)
|
||||
if not next_block:
|
||||
break
|
||||
|
||||
next_para = next_block["paras"].get(f"para_{next_para_id}")
|
||||
if not next_para or next_para.get("is_para_title"):
|
||||
break
|
||||
|
||||
# 合并段落文本
|
||||
curr_para_text = curr_para.get("para_text", "")
|
||||
next_para_text = next_para.get("para_text", "")
|
||||
curr_para["para_text"] = curr_para_text + " " + next_para_text
|
||||
|
||||
# 更新 next_para_location
|
||||
curr_para["next_para_location"] = next_para.get("next_para_location")
|
||||
|
||||
# 将下一个段落文本置为空,表示已被合并
|
||||
next_para["para_text"] = ""
|
||||
|
||||
# 更新 merge_next_para 标记
|
||||
curr_para["merge_next_para"] = next_para.get("merge_next_para", False)
|
||||
|
||||
return pdf_dict
|
||||
@@ -0,0 +1,486 @@
|
||||
import sys
|
||||
|
||||
from libs.commons import fitz
|
||||
|
||||
from termcolor import cprint
|
||||
|
||||
from para.commons import *
|
||||
|
||||
|
||||
if sys.version_info[0] >= 3:
|
||||
sys.stdout.reconfigure(encoding="utf-8") # type: ignore
|
||||
|
||||
|
||||
|
||||
class BlockTerminationProcessor:
|
||||
def __init__(self) -> None:
|
||||
pass
|
||||
|
||||
def _is_consistent_lines(
|
||||
self,
|
||||
curr_line,
|
||||
prev_line,
|
||||
next_line,
|
||||
consistent_direction, # 0 for prev, 1 for next, 2 for both
|
||||
):
|
||||
"""
|
||||
This function checks if the line is consistent with its neighbors
|
||||
|
||||
Parameters
|
||||
----------
|
||||
curr_line : dict
|
||||
current line
|
||||
prev_line : dict
|
||||
previous line
|
||||
next_line : dict
|
||||
next line
|
||||
consistent_direction : int
|
||||
0 for prev, 1 for next, 2 for both
|
||||
|
||||
Returns
|
||||
-------
|
||||
bool
|
||||
True if the line is consistent with its neighbors, False otherwise.
|
||||
"""
|
||||
|
||||
curr_line_font_size = curr_line["spans"][0]["size"]
|
||||
curr_line_font_type = curr_line["spans"][0]["font"].lower()
|
||||
|
||||
if consistent_direction == 0:
|
||||
if prev_line:
|
||||
prev_line_font_size = prev_line["spans"][0]["size"]
|
||||
prev_line_font_type = prev_line["spans"][0]["font"].lower()
|
||||
return curr_line_font_size == prev_line_font_size and curr_line_font_type == prev_line_font_type
|
||||
else:
|
||||
return False
|
||||
|
||||
elif consistent_direction == 1:
|
||||
if next_line:
|
||||
next_line_font_size = next_line["spans"][0]["size"]
|
||||
next_line_font_type = next_line["spans"][0]["font"].lower()
|
||||
return curr_line_font_size == next_line_font_size and curr_line_font_type == next_line_font_type
|
||||
else:
|
||||
return False
|
||||
|
||||
elif consistent_direction == 2:
|
||||
if prev_line and next_line:
|
||||
prev_line_font_size = prev_line["spans"][0]["size"]
|
||||
prev_line_font_type = prev_line["spans"][0]["font"].lower()
|
||||
next_line_font_size = next_line["spans"][0]["size"]
|
||||
next_line_font_type = next_line["spans"][0]["font"].lower()
|
||||
return (curr_line_font_size == prev_line_font_size and curr_line_font_type == prev_line_font_type) and (
|
||||
curr_line_font_size == next_line_font_size and curr_line_font_type == next_line_font_type
|
||||
)
|
||||
else:
|
||||
return False
|
||||
|
||||
else:
|
||||
return False
|
||||
|
||||
def _is_regular_line(self, curr_line_bbox, prev_line_bbox, next_line_bbox, avg_char_width, X0, X1, avg_line_height):
|
||||
"""
|
||||
This function checks if the line is a regular line
|
||||
|
||||
Parameters
|
||||
----------
|
||||
curr_line_bbox : list
|
||||
bbox of the current line
|
||||
prev_line_bbox : list
|
||||
bbox of the previous line
|
||||
next_line_bbox : list
|
||||
bbox of the next line
|
||||
avg_char_width : float
|
||||
average of char widths
|
||||
X0 : float
|
||||
median of x0 values, which represents the left average boundary of the page
|
||||
X1 : float
|
||||
median of x1 values, which represents the right average boundary of the page
|
||||
avg_line_height : float
|
||||
average of line heights
|
||||
|
||||
Returns
|
||||
-------
|
||||
bool
|
||||
True if the line is a regular line, False otherwise.
|
||||
"""
|
||||
horizontal_ratio = 0.5
|
||||
vertical_ratio = 0.5
|
||||
horizontal_thres = horizontal_ratio * avg_char_width
|
||||
vertical_thres = vertical_ratio * avg_line_height
|
||||
|
||||
x0, y0, x1, y1 = curr_line_bbox
|
||||
|
||||
x0_near_X0 = abs(x0 - X0) < horizontal_thres
|
||||
x1_near_X1 = abs(x1 - X1) < horizontal_thres
|
||||
|
||||
prev_line_is_end_of_para = prev_line_bbox and (abs(prev_line_bbox[2] - X1) > avg_char_width)
|
||||
|
||||
sufficient_spacing_above = False
|
||||
if prev_line_bbox:
|
||||
vertical_spacing_above = y1 - prev_line_bbox[3]
|
||||
sufficient_spacing_above = vertical_spacing_above > vertical_thres
|
||||
|
||||
sufficient_spacing_below = False
|
||||
if next_line_bbox:
|
||||
vertical_spacing_below = next_line_bbox[1] - y0
|
||||
sufficient_spacing_below = vertical_spacing_below > vertical_thres
|
||||
|
||||
return (
|
||||
(sufficient_spacing_above or sufficient_spacing_below)
|
||||
or (not x0_near_X0 and not x1_near_X1)
|
||||
or prev_line_is_end_of_para
|
||||
)
|
||||
|
||||
def _is_possible_start_of_para(self, curr_line, prev_line, next_line, X0, X1, avg_char_width, avg_font_size):
|
||||
"""
|
||||
This function checks if the line is a possible start of a paragraph
|
||||
|
||||
Parameters
|
||||
----------
|
||||
curr_line : dict
|
||||
current line
|
||||
prev_line : dict
|
||||
previous line
|
||||
next_line : dict
|
||||
next line
|
||||
X0 : float
|
||||
median of x0 values, which represents the left average boundary of the page
|
||||
X1 : float
|
||||
median of x1 values, which represents the right average boundary of the page
|
||||
avg_char_width : float
|
||||
average of char widths
|
||||
avg_line_height : float
|
||||
average of line heights
|
||||
|
||||
Returns
|
||||
-------
|
||||
bool
|
||||
True if the line is a possible start of a paragraph, False otherwise.
|
||||
"""
|
||||
start_confidence = 0.5 # Initial confidence of the line being a start of a paragraph
|
||||
decision_path = [] # Record the decision path
|
||||
|
||||
curr_line_bbox = curr_line["bbox"]
|
||||
prev_line_bbox = prev_line["bbox"] if prev_line else None
|
||||
next_line_bbox = next_line["bbox"] if next_line else None
|
||||
|
||||
indent_ratio = 1
|
||||
|
||||
vertical_ratio = 1.5
|
||||
vertical_thres = vertical_ratio * avg_font_size
|
||||
|
||||
left_horizontal_ratio = 0.5
|
||||
left_horizontal_thres = left_horizontal_ratio * avg_char_width
|
||||
|
||||
right_horizontal_ratio = 2.5
|
||||
right_horizontal_thres = right_horizontal_ratio * avg_char_width
|
||||
|
||||
x0, y0, x1, y1 = curr_line_bbox
|
||||
|
||||
indent_condition = x0 > X0 + indent_ratio * avg_char_width
|
||||
if indent_condition:
|
||||
start_confidence += 0.2
|
||||
decision_path.append("indent_condition_met")
|
||||
|
||||
x0_near_X0 = abs(x0 - X0) < left_horizontal_thres
|
||||
if x0_near_X0:
|
||||
start_confidence += 0.1
|
||||
decision_path.append("x0_near_X0")
|
||||
|
||||
x1_near_X1 = abs(x1 - X1) < right_horizontal_thres
|
||||
if x1_near_X1:
|
||||
start_confidence += 0.1
|
||||
decision_path.append("x1_near_X1")
|
||||
|
||||
if prev_line is None:
|
||||
prev_line_is_end_of_para = True
|
||||
start_confidence += 0.2
|
||||
decision_path.append("no_prev_line")
|
||||
else:
|
||||
prev_line_is_end_of_para, _, _ = self._is_possible_end_of_para(prev_line, next_line, X0, X1, avg_char_width)
|
||||
if prev_line_is_end_of_para:
|
||||
start_confidence += 0.1
|
||||
decision_path.append("prev_line_is_end_of_para")
|
||||
|
||||
sufficient_spacing_above = False
|
||||
if prev_line_bbox:
|
||||
vertical_spacing_above = y1 - prev_line_bbox[3]
|
||||
sufficient_spacing_above = vertical_spacing_above > vertical_thres
|
||||
if sufficient_spacing_above:
|
||||
start_confidence += 0.2
|
||||
decision_path.append("sufficient_spacing_above")
|
||||
|
||||
sufficient_spacing_below = False
|
||||
if next_line_bbox:
|
||||
vertical_spacing_below = next_line_bbox[1] - y0
|
||||
sufficient_spacing_below = vertical_spacing_below > vertical_thres
|
||||
if sufficient_spacing_below:
|
||||
start_confidence += 0.2
|
||||
decision_path.append("sufficient_spacing_below")
|
||||
|
||||
is_regular_line = self._is_regular_line(
|
||||
curr_line_bbox, prev_line_bbox, next_line_bbox, avg_char_width, X0, X1, avg_font_size
|
||||
)
|
||||
if is_regular_line:
|
||||
start_confidence += 0.1
|
||||
decision_path.append("is_regular_line")
|
||||
|
||||
is_start_of_para = (
|
||||
(sufficient_spacing_above or sufficient_spacing_below)
|
||||
or (indent_condition)
|
||||
or (not indent_condition and x0_near_X0 and x1_near_X1 and not is_regular_line)
|
||||
or prev_line_is_end_of_para
|
||||
)
|
||||
return (is_start_of_para, start_confidence, decision_path)
|
||||
|
||||
def _is_possible_end_of_para(self, curr_line, next_line, X0, X1, avg_char_width):
|
||||
"""
|
||||
This function checks if the line is a possible end of a paragraph
|
||||
|
||||
Parameters
|
||||
----------
|
||||
curr_line : dict
|
||||
current line
|
||||
next_line : dict
|
||||
next line
|
||||
X0 : float
|
||||
median of x0 values, which represents the left average boundary of the page
|
||||
X1 : float
|
||||
median of x1 values, which represents the right average boundary of the page
|
||||
avg_char_width : float
|
||||
average of char widths
|
||||
|
||||
Returns
|
||||
-------
|
||||
bool
|
||||
True if the line is a possible end of a paragraph, False otherwise.
|
||||
"""
|
||||
|
||||
end_confidence = 0.5 # Initial confidence of the line being a end of a paragraph
|
||||
decision_path = [] # Record the decision path
|
||||
|
||||
curr_line_bbox = curr_line["bbox"]
|
||||
next_line_bbox = next_line["bbox"] if next_line else None
|
||||
|
||||
left_horizontal_ratio = 0.5
|
||||
right_horizontal_ratio = 0.5
|
||||
|
||||
x0, _, x1, y1 = curr_line_bbox
|
||||
next_x0, next_y0, _, _ = next_line_bbox if next_line_bbox else (0, 0, 0, 0)
|
||||
|
||||
x0_near_X0 = abs(x0 - X0) < left_horizontal_ratio * avg_char_width
|
||||
if x0_near_X0:
|
||||
end_confidence += 0.1
|
||||
decision_path.append("x0_near_X0")
|
||||
|
||||
x1_smaller_than_X1 = x1 < X1 - right_horizontal_ratio * avg_char_width
|
||||
if x1_smaller_than_X1:
|
||||
end_confidence += 0.1
|
||||
decision_path.append("x1_smaller_than_X1")
|
||||
|
||||
next_line_is_start_of_para = (
|
||||
next_line_bbox
|
||||
and (next_x0 > X0 + left_horizontal_ratio * avg_char_width)
|
||||
and (not is_line_left_aligned_from_neighbors(curr_line_bbox, None, next_line_bbox, avg_char_width, direction=1))
|
||||
)
|
||||
if next_line_is_start_of_para:
|
||||
end_confidence += 0.2
|
||||
decision_path.append("next_line_is_start_of_para")
|
||||
|
||||
is_line_left_aligned_from_neighbors_bool = is_line_left_aligned_from_neighbors(
|
||||
curr_line_bbox, None, next_line_bbox, avg_char_width
|
||||
)
|
||||
if is_line_left_aligned_from_neighbors_bool:
|
||||
end_confidence += 0.1
|
||||
decision_path.append("line_is_left_aligned_from_neighbors")
|
||||
|
||||
is_line_right_aligned_from_neighbors_bool = is_line_right_aligned_from_neighbors(
|
||||
curr_line_bbox, None, next_line_bbox, avg_char_width
|
||||
)
|
||||
if not is_line_right_aligned_from_neighbors_bool:
|
||||
end_confidence += 0.1
|
||||
decision_path.append("line_is_not_right_aligned_from_neighbors")
|
||||
|
||||
is_end_of_para = end_with_punctuation(curr_line["text"]) and (
|
||||
(x0_near_X0 and x1_smaller_than_X1)
|
||||
or (is_line_left_aligned_from_neighbors_bool and not is_line_right_aligned_from_neighbors_bool)
|
||||
)
|
||||
|
||||
return (is_end_of_para, end_confidence, decision_path)
|
||||
|
||||
def _cut_paras_per_block(
|
||||
self,
|
||||
block,
|
||||
):
|
||||
"""
|
||||
Processes a raw block from PyMuPDF and returns the processed block.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
raw_block : dict
|
||||
A raw block from pymupdf.
|
||||
|
||||
Returns
|
||||
-------
|
||||
processed_block : dict
|
||||
|
||||
"""
|
||||
|
||||
def _construct_para(lines, is_block_title, para_title_level):
|
||||
"""
|
||||
Construct a paragraph from given lines.
|
||||
"""
|
||||
|
||||
font_sizes = [span["size"] for line in lines for span in line["spans"]]
|
||||
avg_font_size = sum(font_sizes) / len(font_sizes) if font_sizes else 0
|
||||
|
||||
font_colors = [span["color"] for line in lines for span in line["spans"]]
|
||||
most_common_font_color = max(set(font_colors), key=font_colors.count) if font_colors else None
|
||||
|
||||
# font_types = [span["font"] for line in lines for span in line["spans"]]
|
||||
# most_common_font_type = max(set(font_types), key=font_types.count) if font_types else None
|
||||
|
||||
font_type_lengths = {}
|
||||
for line in lines:
|
||||
for span in line["spans"]:
|
||||
font_type = span["font"]
|
||||
bbox_width = span["bbox"][2] - span["bbox"][0]
|
||||
if font_type in font_type_lengths:
|
||||
font_type_lengths[font_type] += bbox_width
|
||||
else:
|
||||
font_type_lengths[font_type] = bbox_width
|
||||
|
||||
# get the font type with the longest bbox width
|
||||
most_common_font_type = max(font_type_lengths, key=font_type_lengths.get) if font_type_lengths else None # type: ignore
|
||||
|
||||
para_bbox = calculate_para_bbox(lines)
|
||||
para_text = " ".join(line["text"] for line in lines)
|
||||
|
||||
return {
|
||||
"para_bbox": para_bbox,
|
||||
"para_text": para_text,
|
||||
"para_font_type": most_common_font_type,
|
||||
"para_font_size": avg_font_size,
|
||||
"para_font_color": most_common_font_color,
|
||||
"is_para_title": is_block_title,
|
||||
"para_title_level": para_title_level,
|
||||
}
|
||||
|
||||
block_bbox = block["bbox"]
|
||||
block_text = block["text"]
|
||||
block_lines = block["lines"]
|
||||
|
||||
X0 = safe_get(block, "X0", 0)
|
||||
X1 = safe_get(block, "X1", 0)
|
||||
avg_char_width = safe_get(block, "avg_char_width", 0)
|
||||
avg_char_height = safe_get(block, "avg_char_height", 0)
|
||||
avg_font_size = safe_get(block, "avg_font_size", 0)
|
||||
|
||||
is_block_title = safe_get(block, "is_block_title", False)
|
||||
para_title_level = safe_get(block, "block_title_level", 0)
|
||||
|
||||
# Segment into paragraphs
|
||||
para_ranges = []
|
||||
in_paragraph = False
|
||||
start_idx_of_para = None
|
||||
|
||||
# Create the processed paragraphs
|
||||
processed_paras = {}
|
||||
para_bboxes = []
|
||||
end_idx_of_para = 0
|
||||
|
||||
for line_index, line in enumerate(block_lines):
|
||||
curr_line = line
|
||||
prev_line = block_lines[line_index - 1] if line_index > 0 else None
|
||||
next_line = block_lines[line_index + 1] if line_index < len(block_lines) - 1 else None
|
||||
|
||||
"""
|
||||
Start processing paragraphs.
|
||||
"""
|
||||
|
||||
# Check if the line is the start of a paragraph
|
||||
is_start_of_para, start_confidence, decision_path = self._is_possible_start_of_para(
|
||||
curr_line, prev_line, next_line, X0, X1, avg_char_width, avg_font_size
|
||||
)
|
||||
if not in_paragraph and is_start_of_para:
|
||||
in_paragraph = True
|
||||
start_idx_of_para = line_index
|
||||
|
||||
# print_green(">>> Start of a paragraph")
|
||||
# print(" curr_line_text: ", curr_line["text"])
|
||||
# print(" start_confidence: ", start_confidence)
|
||||
# print(" decision_path: ", decision_path)
|
||||
|
||||
# Check if the line is the end of a paragraph
|
||||
is_end_of_para, end_confidence, decision_path = self._is_possible_end_of_para(
|
||||
curr_line, next_line, X0, X1, avg_char_width
|
||||
)
|
||||
if in_paragraph and (is_end_of_para or not next_line):
|
||||
para_ranges.append((start_idx_of_para, line_index))
|
||||
start_idx_of_para = None
|
||||
in_paragraph = False
|
||||
|
||||
# print_red(">>> End of a paragraph")
|
||||
# print(" curr_line_text: ", curr_line["text"])
|
||||
# print(" end_confidence: ", end_confidence)
|
||||
# print(" decision_path: ", decision_path)
|
||||
|
||||
# Add the last paragraph if it is not added
|
||||
if in_paragraph and start_idx_of_para is not None:
|
||||
para_ranges.append((start_idx_of_para, len(block_lines) - 1))
|
||||
|
||||
# Process the matched paragraphs
|
||||
for para_index, (start_idx, end_idx) in enumerate(para_ranges):
|
||||
matched_lines = block_lines[start_idx : end_idx + 1]
|
||||
para_properties = _construct_para(matched_lines, is_block_title, para_title_level)
|
||||
para_key = f"para_{len(processed_paras)}"
|
||||
processed_paras[para_key] = para_properties
|
||||
para_bboxes.append(para_properties["para_bbox"])
|
||||
end_idx_of_para = end_idx + 1
|
||||
|
||||
# Deal with the remaining lines
|
||||
if end_idx_of_para < len(block_lines):
|
||||
unmatched_lines = block_lines[end_idx_of_para:]
|
||||
unmatched_properties = _construct_para(unmatched_lines, is_block_title, para_title_level)
|
||||
unmatched_key = f"para_{len(processed_paras)}"
|
||||
processed_paras[unmatched_key] = unmatched_properties
|
||||
para_bboxes.append(unmatched_properties["para_bbox"])
|
||||
|
||||
block["paras"] = processed_paras
|
||||
|
||||
return block
|
||||
|
||||
def batch_process_blocks(self, pdf_dict):
|
||||
"""
|
||||
Parses the blocks of all pages.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
pdf_dict : dict
|
||||
PDF dictionary.
|
||||
filter_blocks : list
|
||||
List of bounding boxes to filter.
|
||||
|
||||
Returns
|
||||
-------
|
||||
result_dict : dict
|
||||
Result dictionary.
|
||||
|
||||
"""
|
||||
|
||||
num_paras = 0
|
||||
|
||||
for page_id, page in pdf_dict.items():
|
||||
if page_id.startswith("page_"):
|
||||
para_blocks = []
|
||||
if "para_blocks" in page.keys():
|
||||
input_blocks = page["para_blocks"]
|
||||
for input_block in input_blocks:
|
||||
new_block = self._cut_paras_per_block(input_block)
|
||||
para_blocks.append(new_block)
|
||||
num_paras += len(new_block["paras"])
|
||||
|
||||
page["para_blocks"] = para_blocks
|
||||
|
||||
pdf_dict["statistics"]["num_paras"] = num_paras
|
||||
return pdf_dict
|
||||
+222
@@ -0,0 +1,222 @@
|
||||
import sys
|
||||
|
||||
from libs.commons import fitz
|
||||
from termcolor import cprint
|
||||
|
||||
|
||||
if sys.version_info[0] >= 3:
|
||||
sys.stdout.reconfigure(encoding="utf-8") # type: ignore
|
||||
|
||||
|
||||
def open_pdf(pdf_path):
|
||||
try:
|
||||
pdf_document = fitz.open(pdf_path) # type: ignore
|
||||
return pdf_document
|
||||
except Exception as e:
|
||||
print(f"无法打开PDF文件:{pdf_path}。原因是:{e}")
|
||||
raise e
|
||||
|
||||
|
||||
def print_green_on_red(text):
|
||||
cprint(text, "green", "on_red", attrs=["bold"], end="\n\n")
|
||||
|
||||
|
||||
def print_green(text):
|
||||
print()
|
||||
cprint(text, "green", attrs=["bold"], end="\n\n")
|
||||
|
||||
|
||||
def print_red(text):
|
||||
print()
|
||||
cprint(text, "red", attrs=["bold"], end="\n\n")
|
||||
|
||||
|
||||
def print_yellow(text):
|
||||
print()
|
||||
cprint(text, "yellow", attrs=["bold"], end="\n\n")
|
||||
|
||||
|
||||
def safe_get(dict_obj, key, default):
|
||||
val = dict_obj.get(key)
|
||||
if val is None:
|
||||
return default
|
||||
else:
|
||||
return val
|
||||
|
||||
|
||||
def is_bbox_overlap(bbox1, bbox2):
|
||||
"""
|
||||
This function checks if bbox1 and bbox2 overlap or not
|
||||
|
||||
Parameters
|
||||
----------
|
||||
bbox1 : list
|
||||
bbox1
|
||||
bbox2 : list
|
||||
bbox2
|
||||
|
||||
Returns
|
||||
-------
|
||||
bool
|
||||
True if bbox1 and bbox2 overlap, else False
|
||||
"""
|
||||
x0_1, y0_1, x1_1, y1_1 = bbox1
|
||||
x0_2, y0_2, x1_2, y1_2 = bbox2
|
||||
|
||||
if x0_1 > x1_2 or x0_2 > x1_1:
|
||||
return False
|
||||
if y0_1 > y1_2 or y0_2 > y1_1:
|
||||
return False
|
||||
|
||||
return True
|
||||
|
||||
|
||||
def is_in_bbox(bbox1, bbox2):
|
||||
"""
|
||||
This function checks if bbox1 is in bbox2
|
||||
|
||||
Parameters
|
||||
----------
|
||||
bbox1 : list
|
||||
bbox1
|
||||
bbox2 : list
|
||||
bbox2
|
||||
|
||||
Returns
|
||||
-------
|
||||
bool
|
||||
True if bbox1 is in bbox2, else False
|
||||
"""
|
||||
x0_1, y0_1, x1_1, y1_1 = bbox1
|
||||
x0_2, y0_2, x1_2, y1_2 = bbox2
|
||||
|
||||
if x0_1 >= x0_2 and y0_1 >= y0_2 and x1_1 <= x1_2 and y1_1 <= y1_2:
|
||||
return True
|
||||
else:
|
||||
return False
|
||||
|
||||
|
||||
def calculate_para_bbox(lines):
|
||||
"""
|
||||
This function calculates the minimum bbox of the paragraph
|
||||
|
||||
Parameters
|
||||
----------
|
||||
lines : list
|
||||
lines
|
||||
|
||||
Returns
|
||||
-------
|
||||
para_bbox : list
|
||||
bbox of the paragraph
|
||||
"""
|
||||
x0 = min(line["bbox"][0] for line in lines)
|
||||
y0 = min(line["bbox"][1] for line in lines)
|
||||
x1 = max(line["bbox"][2] for line in lines)
|
||||
y1 = max(line["bbox"][3] for line in lines)
|
||||
return [x0, y0, x1, y1]
|
||||
|
||||
|
||||
def is_line_right_aligned_from_neighbors(curr_line_bbox, prev_line_bbox, next_line_bbox, avg_char_width, direction=2):
|
||||
"""
|
||||
This function checks if the line is right aligned from its neighbors
|
||||
|
||||
Parameters
|
||||
----------
|
||||
curr_line_bbox : list
|
||||
bbox of the current line
|
||||
prev_line_bbox : list
|
||||
bbox of the previous line
|
||||
next_line_bbox : list
|
||||
bbox of the next line
|
||||
avg_char_width : float
|
||||
average of char widths
|
||||
direction : int
|
||||
0 for prev, 1 for next, 2 for both
|
||||
|
||||
Returns
|
||||
-------
|
||||
bool
|
||||
True if the line is right aligned from its neighbors, False otherwise.
|
||||
"""
|
||||
horizontal_ratio = 0.5
|
||||
horizontal_thres = horizontal_ratio * avg_char_width
|
||||
|
||||
_, _, x1, _ = curr_line_bbox
|
||||
_, _, prev_x1, _ = prev_line_bbox if prev_line_bbox else (0, 0, 0, 0)
|
||||
_, _, next_x1, _ = next_line_bbox if next_line_bbox else (0, 0, 0, 0)
|
||||
|
||||
if direction == 0:
|
||||
return abs(x1 - prev_x1) < horizontal_thres
|
||||
elif direction == 1:
|
||||
return abs(x1 - next_x1) < horizontal_thres
|
||||
elif direction == 2:
|
||||
return abs(x1 - prev_x1) < horizontal_thres and abs(x1 - next_x1) < horizontal_thres
|
||||
else:
|
||||
return False
|
||||
|
||||
|
||||
def is_line_left_aligned_from_neighbors(curr_line_bbox, prev_line_bbox, next_line_bbox, avg_char_width, direction=2):
|
||||
"""
|
||||
This function checks if the line is left aligned from its neighbors
|
||||
|
||||
Parameters
|
||||
----------
|
||||
curr_line_bbox : list
|
||||
bbox of the current line
|
||||
prev_line_bbox : list
|
||||
bbox of the previous line
|
||||
next_line_bbox : list
|
||||
bbox of the next line
|
||||
avg_char_width : float
|
||||
average of char widths
|
||||
direction : int
|
||||
0 for prev, 1 for next, 2 for both
|
||||
|
||||
Returns
|
||||
-------
|
||||
bool
|
||||
True if the line is left aligned from its neighbors, False otherwise.
|
||||
"""
|
||||
horizontal_ratio = 0.5
|
||||
horizontal_thres = horizontal_ratio * avg_char_width
|
||||
|
||||
x0, _, _, _ = curr_line_bbox
|
||||
prev_x0, _, _, _ = prev_line_bbox if prev_line_bbox else (0, 0, 0, 0)
|
||||
next_x0, _, _, _ = next_line_bbox if next_line_bbox else (0, 0, 0, 0)
|
||||
|
||||
if direction == 0:
|
||||
return abs(x0 - prev_x0) < horizontal_thres
|
||||
elif direction == 1:
|
||||
return abs(x0 - next_x0) < horizontal_thres
|
||||
elif direction == 2:
|
||||
return abs(x0 - prev_x0) < horizontal_thres and abs(x0 - next_x0) < horizontal_thres
|
||||
else:
|
||||
return False
|
||||
|
||||
|
||||
def end_with_punctuation(line_text):
|
||||
"""
|
||||
This function checks if the line ends with punctuation marks
|
||||
"""
|
||||
|
||||
english_end_puncs = [".", "?", "!"]
|
||||
chinese_end_puncs = ["。", "?", "!"]
|
||||
end_puncs = english_end_puncs + chinese_end_puncs
|
||||
|
||||
last_non_space_char = None
|
||||
for ch in line_text[::-1]:
|
||||
if not ch.isspace():
|
||||
last_non_space_char = ch
|
||||
break
|
||||
|
||||
if last_non_space_char is None:
|
||||
return False
|
||||
|
||||
return last_non_space_char in end_puncs
|
||||
|
||||
|
||||
def is_nested_list(lst):
|
||||
if isinstance(lst, list):
|
||||
return any(isinstance(sub, list) for sub in lst)
|
||||
return False
|
||||
+247
@@ -0,0 +1,247 @@
|
||||
import sys
|
||||
import math
|
||||
|
||||
from collections import defaultdict
|
||||
from para.commons import *
|
||||
|
||||
if sys.version_info[0] >= 3:
|
||||
sys.stdout.reconfigure(encoding="utf-8") # type: ignore
|
||||
|
||||
|
||||
class HeaderFooterProcessor:
|
||||
def __init__(self) -> None:
|
||||
pass
|
||||
|
||||
def get_most_common_bboxes(self, bboxes, page_height, position="top", threshold=0.25, num_bboxes=3, min_frequency=2):
|
||||
"""
|
||||
This function gets the most common bboxes from the bboxes
|
||||
|
||||
Parameters
|
||||
----------
|
||||
bboxes : list
|
||||
bboxes
|
||||
page_height : float
|
||||
height of the page
|
||||
position : str, optional
|
||||
"top" or "bottom", by default "top"
|
||||
threshold : float, optional
|
||||
threshold, by default 0.25
|
||||
num_bboxes : int, optional
|
||||
number of bboxes to return, by default 3
|
||||
min_frequency : int, optional
|
||||
minimum frequency of the bbox, by default 2
|
||||
|
||||
Returns
|
||||
-------
|
||||
common_bboxes : list
|
||||
common bboxes
|
||||
"""
|
||||
# Filter bbox by position
|
||||
if position == "top":
|
||||
filtered_bboxes = [bbox for bbox in bboxes if bbox[1] < page_height * threshold]
|
||||
else:
|
||||
filtered_bboxes = [bbox for bbox in bboxes if bbox[3] > page_height * (1 - threshold)]
|
||||
|
||||
# Find the most common bbox
|
||||
bbox_count = defaultdict(int)
|
||||
for bbox in filtered_bboxes:
|
||||
bbox_count[tuple(bbox)] += 1
|
||||
|
||||
# Get the most frequently occurring bbox, but only consider it when the frequency exceeds min_frequency
|
||||
common_bboxes = [
|
||||
bbox for bbox, count in sorted(bbox_count.items(), key=lambda item: item[1], reverse=True) if count >= min_frequency
|
||||
][:num_bboxes]
|
||||
return common_bboxes
|
||||
|
||||
def detect_footer_header(self, result_dict, similarity_threshold=0.5):
|
||||
"""
|
||||
This function detects the header and footer of the document.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
result_dict : dict
|
||||
result dictionary
|
||||
|
||||
Returns
|
||||
-------
|
||||
result_dict : dict
|
||||
result dictionary
|
||||
"""
|
||||
|
||||
def compare_bbox_with_list(bbox, bbox_list, tolerance=1):
|
||||
return any(all(abs(a - b) < tolerance for a, b in zip(bbox, common_bbox)) for common_bbox in bbox_list)
|
||||
|
||||
def is_single_line_block(block):
|
||||
# Determine based on the width and height of the block
|
||||
block_width = block["X1"] - block["X0"]
|
||||
block_height = block["bbox"][3] - block["bbox"][1]
|
||||
|
||||
# If the height of the block is close to the average character height and the width is large, it is considered a single line
|
||||
return block_height <= block["avg_char_height"] * 3 and block_width > block["avg_char_width"] * 3
|
||||
|
||||
# Traverse all blocks in the document
|
||||
single_preproc_blocks = 0
|
||||
total_blocks = 0
|
||||
single_preproc_blocks = 0
|
||||
|
||||
for page_id, blocks in result_dict.items():
|
||||
if page_id.startswith("page_"):
|
||||
for block_key, block in blocks.items():
|
||||
if block_key.startswith("block_"):
|
||||
total_blocks += 1
|
||||
if is_single_line_block(block):
|
||||
single_preproc_blocks += 1
|
||||
|
||||
# If there are no blocks, skip the header and footer detection
|
||||
if total_blocks == 0:
|
||||
print("No blocks found. Skipping header/footer detection.")
|
||||
return result_dict
|
||||
|
||||
# If most of the blocks are single-line, skip the header and footer detection
|
||||
if single_preproc_blocks / total_blocks > 0.5: # 50% of the blocks are single-line
|
||||
return result_dict
|
||||
|
||||
# Collect the bounding boxes of all blocks
|
||||
all_bboxes = []
|
||||
all_texts = []
|
||||
|
||||
for page_id, blocks in result_dict.items():
|
||||
if page_id.startswith("page_"):
|
||||
for block_key, block in blocks.items():
|
||||
if block_key.startswith("block_"):
|
||||
all_bboxes.append(block["bbox"])
|
||||
|
||||
# Get the height of the page
|
||||
page_height = max(bbox[3] for bbox in all_bboxes)
|
||||
|
||||
# Get the most common bbox lists for headers and footers
|
||||
common_header_bboxes = self.get_most_common_bboxes(all_bboxes, page_height, position="top") if all_bboxes else []
|
||||
common_footer_bboxes = self.get_most_common_bboxes(all_bboxes, page_height, position="bottom") if all_bboxes else []
|
||||
|
||||
# Detect and mark headers and footers
|
||||
for page_id, blocks in result_dict.items():
|
||||
if page_id.startswith("page_"):
|
||||
for block_key, block in blocks.items():
|
||||
if block_key.startswith("block_"):
|
||||
bbox = block["bbox"]
|
||||
text = block["text"]
|
||||
|
||||
is_header = compare_bbox_with_list(bbox, common_header_bboxes)
|
||||
is_footer = compare_bbox_with_list(bbox, common_footer_bboxes)
|
||||
|
||||
block["is_header"] = int(is_header)
|
||||
block["is_footer"] = int(is_footer)
|
||||
|
||||
return result_dict
|
||||
|
||||
|
||||
class NonHorizontalTextProcessor:
|
||||
def __init__(self) -> None:
|
||||
pass
|
||||
|
||||
def detect_non_horizontal_texts(self, result_dict):
|
||||
"""
|
||||
This function detects watermarks and vertical margin notes in the document.
|
||||
|
||||
Watermarks are identified by finding blocks with the same coordinates and frequently occurring identical texts across multiple pages.
|
||||
If these conditions are met, the blocks are highly likely to be watermarks, as opposed to headers or footers, which can change from page to page.
|
||||
If the direction of these blocks is not horizontal, they are definitely considered to be watermarks.
|
||||
|
||||
Vertical margin notes are identified by finding blocks with the same coordinates and frequently occurring identical texts across multiple pages.
|
||||
If these conditions are met, the blocks are highly likely to be vertical margin notes, which typically appear on the left and right sides of the page.
|
||||
If the direction of these blocks is vertical, they are definitely considered to be vertical margin notes.
|
||||
|
||||
|
||||
Parameters
|
||||
----------
|
||||
result_dict : dict
|
||||
The result dictionary.
|
||||
|
||||
Returns
|
||||
-------
|
||||
result_dict : dict
|
||||
The updated result dictionary.
|
||||
"""
|
||||
# Dictionary to store information about potential watermarks
|
||||
potential_watermarks = {}
|
||||
potential_margin_notes = {}
|
||||
|
||||
for page_id, page_content in result_dict.items():
|
||||
if page_id.startswith("page_"):
|
||||
for block_id, block_data in page_content.items():
|
||||
if block_id.startswith("block_"):
|
||||
if "dir" in block_data:
|
||||
coordinates_text = (block_data["bbox"], block_data["text"]) # Tuple of coordinates and text
|
||||
|
||||
angle = math.atan2(block_data["dir"][1], block_data["dir"][0])
|
||||
angle = abs(math.degrees(angle))
|
||||
|
||||
if angle > 5 and angle < 85: # Check if direction is watermarks
|
||||
if coordinates_text in potential_watermarks:
|
||||
potential_watermarks[coordinates_text] += 1
|
||||
else:
|
||||
potential_watermarks[coordinates_text] = 1
|
||||
|
||||
if angle > 85 and angle < 105: # Check if direction is vertical
|
||||
if coordinates_text in potential_margin_notes:
|
||||
potential_margin_notes[coordinates_text] += 1 # Increment count
|
||||
else:
|
||||
potential_margin_notes[coordinates_text] = 1 # Initialize count
|
||||
|
||||
# Identify watermarks by finding entries with counts higher than a threshold (e.g., appearing on more than half of the pages)
|
||||
watermark_threshold = len(result_dict) // 2
|
||||
watermarks = {k: v for k, v in potential_watermarks.items() if v > watermark_threshold}
|
||||
|
||||
# Identify margin notes by finding entries with counts higher than a threshold (e.g., appearing on more than half of the pages)
|
||||
margin_note_threshold = len(result_dict) // 2
|
||||
margin_notes = {k: v for k, v in potential_margin_notes.items() if v > margin_note_threshold}
|
||||
|
||||
# Add watermark information to the result dictionary
|
||||
for page_id, blocks in result_dict.items():
|
||||
if page_id.startswith("page_"):
|
||||
for block_id, block_data in blocks.items():
|
||||
coordinates_text = (block_data["bbox"], block_data["text"])
|
||||
if coordinates_text in watermarks:
|
||||
block_data["is_watermark"] = 1
|
||||
else:
|
||||
block_data["is_watermark"] = 0
|
||||
|
||||
if coordinates_text in margin_notes:
|
||||
block_data["is_vertical_margin_note"] = 1
|
||||
else:
|
||||
block_data["is_vertical_margin_note"] = 0
|
||||
|
||||
return result_dict
|
||||
|
||||
|
||||
class NoiseRemover:
|
||||
def __init__(self) -> None:
|
||||
pass
|
||||
|
||||
def skip_data_noises(self, result_dict):
|
||||
"""
|
||||
This function skips the data noises, including overlap blocks, header, footer, watermark, vertical margin note, title
|
||||
"""
|
||||
filtered_result_dict = {}
|
||||
for page_id, blocks in result_dict.items():
|
||||
if page_id.startswith("page_"):
|
||||
filtered_blocks = {}
|
||||
for block_id, block in blocks.items():
|
||||
if block_id.startswith("block_"):
|
||||
if any(
|
||||
block.get(key, 0)
|
||||
for key in [
|
||||
"is_overlap",
|
||||
"is_header",
|
||||
"is_footer",
|
||||
"is_watermark",
|
||||
"is_vertical_margin_note",
|
||||
"is_block_title",
|
||||
]
|
||||
):
|
||||
continue
|
||||
filtered_blocks[block_id] = block
|
||||
if filtered_blocks:
|
||||
filtered_result_dict[page_id] = filtered_blocks
|
||||
|
||||
return filtered_result_dict
|
||||
+123
@@ -0,0 +1,123 @@
|
||||
import sys
|
||||
|
||||
from libs.commons import fitz
|
||||
|
||||
from para.commons import *
|
||||
|
||||
|
||||
if sys.version_info[0] >= 3:
|
||||
sys.stdout.reconfigure(encoding="utf-8") # type: ignore
|
||||
|
||||
|
||||
class DrawAnnos:
|
||||
"""
|
||||
This class draws annotations on the pdf file
|
||||
|
||||
----------------------------------------
|
||||
Color Code
|
||||
----------------------------------------
|
||||
Red: (1, 0, 0)
|
||||
Green: (0, 1, 0)
|
||||
Blue: (0, 0, 1)
|
||||
Yellow: (1, 1, 0) - mix of red and green
|
||||
Cyan: (0, 1, 1) - mix of green and blue
|
||||
Magenta: (1, 0, 1) - mix of red and blue
|
||||
White: (1, 1, 1) - red, green and blue full intensity
|
||||
Black: (0, 0, 0) - no color component whatsoever
|
||||
Gray: (0.5, 0.5, 0.5) - equal and medium intensity of red, green and blue color components
|
||||
Orange: (1, 0.65, 0) - maximum intensity of red, medium intensity of green, no blue component
|
||||
"""
|
||||
|
||||
def __init__(self) -> None:
|
||||
pass
|
||||
|
||||
def __is_nested_list(self, lst):
|
||||
"""
|
||||
This function returns True if the given list is a nested list of any degree.
|
||||
"""
|
||||
if isinstance(lst, list):
|
||||
return any(self.__is_nested_list(i) for i in lst) or any(isinstance(i, list) for i in lst)
|
||||
return False
|
||||
|
||||
def __valid_rect(self, bbox):
|
||||
# Ensure that the rectangle is not empty or invalid
|
||||
if isinstance(bbox[0], list):
|
||||
return False # It's a nested list, hence it can't be valid rect
|
||||
else:
|
||||
return bbox[0] < bbox[2] and bbox[1] < bbox[3]
|
||||
|
||||
def __draw_nested_boxes(self, page, nested_bbox, color=(0, 1, 1)):
|
||||
"""
|
||||
This function draws the nested boxes
|
||||
|
||||
Parameters
|
||||
----------
|
||||
page : fitz.Page
|
||||
page
|
||||
nested_bbox : list
|
||||
nested bbox
|
||||
color : tuple
|
||||
color, by default (0, 1, 1) # draw with cyan color for combined paragraph
|
||||
"""
|
||||
if self.__is_nested_list(nested_bbox): # If it's a nested list
|
||||
for bbox in nested_bbox:
|
||||
self.__draw_nested_boxes(page, bbox, color) # Recursively call the function
|
||||
elif self.__valid_rect(nested_bbox): # If valid rectangle
|
||||
para_rect = fitz.Rect(nested_bbox)
|
||||
para_anno = page.add_rect_annot(para_rect)
|
||||
para_anno.set_colors(stroke=color) # draw with cyan color for combined paragraph
|
||||
para_anno.set_border(width=1)
|
||||
para_anno.update()
|
||||
|
||||
def draw_annos(self, input_pdf_path, pdf_dic, output_pdf_path):
|
||||
pdf_doc = open_pdf(input_pdf_path)
|
||||
|
||||
if pdf_dic is None:
|
||||
pdf_dic = {}
|
||||
|
||||
if output_pdf_path is None:
|
||||
output_pdf_path = input_pdf_path.replace(".pdf", "_anno.pdf")
|
||||
|
||||
for page_id, page in enumerate(pdf_doc): # type: ignore
|
||||
page_key = f"page_{page_id}"
|
||||
for ele_key, ele_data in pdf_dic[page_key].items():
|
||||
if ele_key == "para_blocks":
|
||||
para_blocks = ele_data
|
||||
for para_block in para_blocks:
|
||||
if "paras" in para_block.keys():
|
||||
paras = para_block["paras"]
|
||||
for para_key, para_content in paras.items():
|
||||
para_bbox = para_content["para_bbox"]
|
||||
# print(f"para_bbox: {para_bbox}")
|
||||
# print(f"is a nested list: {self.__is_nested_list(para_bbox)}")
|
||||
if self.__is_nested_list(para_bbox) and len(para_bbox) > 1:
|
||||
color = (0, 1, 1)
|
||||
self.__draw_nested_boxes(
|
||||
page, para_bbox, color
|
||||
) # draw with cyan color for combined paragraph
|
||||
else:
|
||||
if self.__valid_rect(para_bbox):
|
||||
para_rect = fitz.Rect(para_bbox)
|
||||
para_anno = page.add_rect_annot(para_rect)
|
||||
para_anno.set_colors(stroke=(0, 1, 0)) # draw with green color for normal paragraph
|
||||
para_anno.set_border(width=0.5)
|
||||
para_anno.update()
|
||||
|
||||
is_para_title = para_content["is_para_title"]
|
||||
if is_para_title:
|
||||
if self.__is_nested_list(para_content["para_bbox"]) and len(para_content["para_bbox"]) > 1:
|
||||
color = (0, 0, 1)
|
||||
self.__draw_nested_boxes(
|
||||
page, para_content["para_bbox"], color
|
||||
) # draw with cyan color for combined title
|
||||
else:
|
||||
if self.__valid_rect(para_content["para_bbox"]):
|
||||
para_rect = fitz.Rect(para_content["para_bbox"])
|
||||
if self.__valid_rect(para_content["para_bbox"]):
|
||||
para_anno = page.add_rect_annot(para_rect)
|
||||
para_anno.set_colors(stroke=(0, 0, 1)) # draw with blue color for normal title
|
||||
para_anno.set_border(width=0.5)
|
||||
para_anno.update()
|
||||
|
||||
pdf_doc.save(output_pdf_path)
|
||||
pdf_doc.close()
|
||||
@@ -0,0 +1,198 @@
|
||||
class DenseSingleLineBlockException(Exception):
|
||||
"""
|
||||
This class defines the exception type for dense single line-block.
|
||||
"""
|
||||
|
||||
def __init__(self, message="DenseSingleLineBlockException"):
|
||||
self.message = message
|
||||
super().__init__(self.message)
|
||||
|
||||
def __str__(self):
|
||||
return f"{self.message}"
|
||||
|
||||
def __repr__(self):
|
||||
return f"{self.message}"
|
||||
|
||||
|
||||
class TitleDetectionException(Exception):
|
||||
"""
|
||||
This class defines the exception type for title detection.
|
||||
"""
|
||||
|
||||
def __init__(self, message="TitleDetectionException"):
|
||||
self.message = message
|
||||
super().__init__(self.message)
|
||||
|
||||
def __str__(self):
|
||||
return f"{self.message}"
|
||||
|
||||
def __repr__(self):
|
||||
return f"{self.message}"
|
||||
|
||||
|
||||
class TitleLevelException(Exception):
|
||||
"""
|
||||
This class defines the exception type for title level.
|
||||
"""
|
||||
|
||||
def __init__(self, message="TitleLevelException"):
|
||||
self.message = message
|
||||
super().__init__(self.message)
|
||||
|
||||
def __str__(self):
|
||||
return f"{self.message}"
|
||||
|
||||
def __repr__(self):
|
||||
return f"{self.message}"
|
||||
|
||||
|
||||
class ParaSplitException(Exception):
|
||||
"""
|
||||
This class defines the exception type for paragraph splitting.
|
||||
"""
|
||||
|
||||
def __init__(self, message="ParaSplitException"):
|
||||
self.message = message
|
||||
super().__init__(self.message)
|
||||
|
||||
def __str__(self):
|
||||
return f"{self.message}"
|
||||
|
||||
def __repr__(self):
|
||||
return f"{self.message}"
|
||||
|
||||
|
||||
class ParaMergeException(Exception):
|
||||
"""
|
||||
This class defines the exception type for paragraph merging.
|
||||
"""
|
||||
|
||||
def __init__(self, message="ParaMergeException"):
|
||||
self.message = message
|
||||
super().__init__(self.message)
|
||||
|
||||
def __str__(self):
|
||||
return f"{self.message}"
|
||||
|
||||
def __repr__(self):
|
||||
return f"{self.message}"
|
||||
|
||||
|
||||
class DiscardByException:
|
||||
"""
|
||||
This class discards pdf files by exception
|
||||
"""
|
||||
|
||||
def __init__(self) -> None:
|
||||
pass
|
||||
|
||||
def discard_by_single_line_block(self, pdf_dic, exception: DenseSingleLineBlockException):
|
||||
"""
|
||||
This function discards pdf files by single line block exception
|
||||
|
||||
Parameters
|
||||
----------
|
||||
pdf_dic : dict
|
||||
pdf dictionary
|
||||
exception : str
|
||||
exception message
|
||||
|
||||
Returns
|
||||
-------
|
||||
error_message : str
|
||||
"""
|
||||
exception_page_nums = 0
|
||||
page_num = 0
|
||||
for page_id, page in pdf_dic.items():
|
||||
if page_id.startswith("page_"):
|
||||
page_num += 1
|
||||
if "preproc_blocks" in page.keys():
|
||||
preproc_blocks = page["preproc_blocks"]
|
||||
|
||||
all_single_line_blocks = []
|
||||
for block in preproc_blocks:
|
||||
if len(block["lines"]) == 1:
|
||||
all_single_line_blocks.append(block)
|
||||
|
||||
if len(preproc_blocks) > 0 and len(all_single_line_blocks) / len(preproc_blocks) > 0.9:
|
||||
exception_page_nums += 1
|
||||
|
||||
if page_num == 0:
|
||||
return None
|
||||
|
||||
if exception_page_nums / page_num > 0.1: # Low ratio means basically, whenever this is the case, it is discarded
|
||||
return exception.message
|
||||
|
||||
return None
|
||||
|
||||
def discard_by_title_detection(self, pdf_dic, exception: TitleDetectionException):
|
||||
"""
|
||||
This function discards pdf files by title detection exception
|
||||
|
||||
Parameters
|
||||
----------
|
||||
pdf_dic : dict
|
||||
pdf dictionary
|
||||
exception : str
|
||||
exception message
|
||||
|
||||
Returns
|
||||
-------
|
||||
error_message : str
|
||||
"""
|
||||
# return exception.message
|
||||
return None
|
||||
|
||||
def discard_by_title_level(self, pdf_dic, exception: TitleLevelException):
|
||||
"""
|
||||
This function discards pdf files by title level exception
|
||||
|
||||
Parameters
|
||||
----------
|
||||
pdf_dic : dict
|
||||
pdf dictionary
|
||||
exception : str
|
||||
exception message
|
||||
|
||||
Returns
|
||||
-------
|
||||
error_message : str
|
||||
"""
|
||||
# return exception.message
|
||||
return None
|
||||
|
||||
def discard_by_split_para(self, pdf_dic, exception: ParaSplitException):
|
||||
"""
|
||||
This function discards pdf files by split para exception
|
||||
|
||||
Parameters
|
||||
----------
|
||||
pdf_dic : dict
|
||||
pdf dictionary
|
||||
exception : str
|
||||
exception message
|
||||
|
||||
Returns
|
||||
-------
|
||||
error_message : str
|
||||
"""
|
||||
# return exception.message
|
||||
return None
|
||||
|
||||
def discard_by_merge_para(self, pdf_dic, exception: ParaMergeException):
|
||||
"""
|
||||
This function discards pdf files by merge para exception
|
||||
|
||||
Parameters
|
||||
----------
|
||||
pdf_dic : dict
|
||||
pdf dictionary
|
||||
exception : str
|
||||
exception message
|
||||
|
||||
Returns
|
||||
-------
|
||||
error_message : str
|
||||
"""
|
||||
# return exception.message
|
||||
return None
|
||||
@@ -0,0 +1,41 @@
|
||||
import sys
|
||||
import math
|
||||
from para.commons import *
|
||||
|
||||
|
||||
if sys.version_info[0] >= 3:
|
||||
sys.stdout.reconfigure(encoding="utf-8") # type: ignore
|
||||
|
||||
|
||||
class LayoutFilterProcessor:
|
||||
def __init__(self) -> None:
|
||||
pass
|
||||
|
||||
def batch_process_blocks(self, pdf_dict):
|
||||
for page_id, blocks in pdf_dict.items():
|
||||
if page_id.startswith("page_"):
|
||||
if "layout_bboxes" in blocks.keys() and "para_blocks" in blocks.keys():
|
||||
layout_bbox_objs = blocks["layout_bboxes"]
|
||||
if layout_bbox_objs is None:
|
||||
continue
|
||||
layout_bboxes = [bbox_obj["layout_bbox"] for bbox_obj in layout_bbox_objs]
|
||||
|
||||
# Use math.ceil function to enlarge each value of x0, y0, x1, y1 of each layout_bbox
|
||||
layout_bboxes = [
|
||||
[math.ceil(x0), math.ceil(y0), math.ceil(x1), math.ceil(y1)] for x0, y0, x1, y1 in layout_bboxes
|
||||
]
|
||||
|
||||
para_blocks = blocks["para_blocks"]
|
||||
if para_blocks is None:
|
||||
continue
|
||||
|
||||
for lb_bbox in layout_bboxes:
|
||||
for i, para_block in enumerate(para_blocks):
|
||||
para_bbox = para_block["bbox"]
|
||||
para_blocks[i]["in_layout"] = 0
|
||||
if is_in_bbox(para_bbox, lb_bbox):
|
||||
para_blocks[i]["in_layout"] = 1
|
||||
|
||||
blocks["para_blocks"] = para_blocks
|
||||
|
||||
return pdf_dict
|
||||
@@ -0,0 +1,298 @@
|
||||
import os
|
||||
import sys
|
||||
import json
|
||||
|
||||
from para.commons import *
|
||||
|
||||
from para.raw_processor import RawBlockProcessor
|
||||
from para.layout_match_processor import LayoutFilterProcessor
|
||||
from para.stats import BlockStatisticsCalculator
|
||||
from para.stats import DocStatisticsCalculator
|
||||
from para.title_processor import TitleProcessor
|
||||
from para.block_termination_processor import BlockTerminationProcessor
|
||||
from para.block_continuation_processor import BlockContinuationProcessor
|
||||
from para.draw import DrawAnnos
|
||||
from para.exceptions import (
|
||||
DenseSingleLineBlockException,
|
||||
TitleDetectionException,
|
||||
TitleLevelException,
|
||||
ParaSplitException,
|
||||
ParaMergeException,
|
||||
DiscardByException,
|
||||
)
|
||||
|
||||
|
||||
if sys.version_info[0] >= 3:
|
||||
sys.stdout.reconfigure(encoding="utf-8") # type: ignore
|
||||
|
||||
|
||||
class ParaProcessPipeline:
|
||||
def __init__(self) -> None:
|
||||
pass
|
||||
|
||||
def para_process_pipeline(self, pdf_info_dict, para_debug_mode=None, input_pdf_path=None, output_pdf_path=None):
|
||||
"""
|
||||
This function processes the paragraphs, including:
|
||||
1. Read raw input json file into pdf_dic
|
||||
2. Detect and replace equations
|
||||
3. Combine spans into a natural line
|
||||
4. Check if the paragraphs are inside bboxes passed from "layout_bboxes" key
|
||||
5. Compute statistics for each block
|
||||
6. Detect titles in the document
|
||||
7. Detect paragraphs inside each block
|
||||
8. Divide the level of the titles
|
||||
9. Detect and combine paragraphs from different blocks into one paragraph
|
||||
10. Check whether the final results after checking headings, dividing paragraphs within blocks, and merging paragraphs between blocks are plausible and reasonable.
|
||||
11. Draw annotations on the pdf file
|
||||
|
||||
Parameters
|
||||
----------
|
||||
pdf_dic_json_fpath : str
|
||||
path to the pdf dictionary json file.
|
||||
Notice: data noises, including overlap blocks, header, footer, watermark, vertical margin note have been removed already.
|
||||
input_pdf_doc : str
|
||||
path to the input pdf file
|
||||
output_pdf_path : str
|
||||
path to the output pdf file
|
||||
|
||||
Returns
|
||||
-------
|
||||
pdf_dict : dict
|
||||
result dictionary
|
||||
"""
|
||||
|
||||
error_info = None
|
||||
|
||||
output_json_file = ""
|
||||
output_dir = ""
|
||||
|
||||
if input_pdf_path is not None:
|
||||
input_pdf_path = os.path.abspath(input_pdf_path)
|
||||
|
||||
# print_green_on_red(f">>>>>>>>>>>>>>>>>>> Process the paragraphs of {input_pdf_path}")
|
||||
|
||||
if output_pdf_path is not None:
|
||||
output_dir = os.path.dirname(output_pdf_path)
|
||||
output_json_file = f"{output_dir}/pdf_dic.json"
|
||||
|
||||
def __save_pdf_dic(pdf_dic, output_pdf_path, stage="0", para_debug_mode=para_debug_mode):
|
||||
"""
|
||||
Save the pdf_dic to a json file
|
||||
"""
|
||||
output_pdf_file_name = os.path.basename(output_pdf_path)
|
||||
# output_dir = os.path.dirname(output_pdf_path)
|
||||
output_dir = "\\tmp\\pdf_parse"
|
||||
output_pdf_file_name = output_pdf_file_name.replace(".pdf", f"_stage_{stage}.json")
|
||||
pdf_dic_json_fpath = os.path.join(output_dir, output_pdf_file_name)
|
||||
|
||||
if not os.path.exists(output_dir):
|
||||
os.makedirs(output_dir)
|
||||
|
||||
if para_debug_mode == "full":
|
||||
with open(pdf_dic_json_fpath, "w", encoding="utf-8") as f:
|
||||
json.dump(pdf_dic, f, indent=2, ensure_ascii=False)
|
||||
|
||||
# Validate the output already exists
|
||||
if not os.path.exists(pdf_dic_json_fpath):
|
||||
print_red(f"Failed to save the pdf_dic to {pdf_dic_json_fpath}")
|
||||
return None
|
||||
else:
|
||||
print_green(f"Succeed to save the pdf_dic to {pdf_dic_json_fpath}")
|
||||
|
||||
return pdf_dic_json_fpath
|
||||
|
||||
"""
|
||||
Preprocess the lines of block
|
||||
"""
|
||||
# Find and replace the interline and inline equations, should be better done before the paragraph processing
|
||||
# Create "para_blocks" for each page.
|
||||
# equationProcessor = EquationsProcessor()
|
||||
# pdf_dic = equationProcessor.batch_process_blocks(pdf_info_dict)
|
||||
|
||||
# Combine spans into a natural line
|
||||
rawBlockProcessor = RawBlockProcessor()
|
||||
pdf_dic = rawBlockProcessor.batch_process_blocks(pdf_info_dict)
|
||||
# print(f"pdf_dic['page_0']['para_blocks'][0]: {pdf_dic['page_0']['para_blocks'][0]}", end="\n\n")
|
||||
|
||||
# Check if the paragraphs are inside bboxes passed from "layout_bboxes" key
|
||||
layoutFilter = LayoutFilterProcessor()
|
||||
pdf_dic = layoutFilter.batch_process_blocks(pdf_dic)
|
||||
|
||||
# Compute statistics for each block
|
||||
blockStatisticsCalculator = BlockStatisticsCalculator()
|
||||
pdf_dic = blockStatisticsCalculator.batch_process_blocks(pdf_dic)
|
||||
# print(f"pdf_dic['page_0']['para_blocks'][0]: {pdf_dic['page_0']['para_blocks'][0]}", end="\n\n")
|
||||
|
||||
# Compute statistics for all blocks(namely this pdf document)
|
||||
docStatisticsCalculator = DocStatisticsCalculator()
|
||||
pdf_dic = docStatisticsCalculator.calc_stats_of_doc(pdf_dic)
|
||||
# print(f"pdf_dic['statistics']: {pdf_dic['statistics']}", end="\n\n")
|
||||
|
||||
# Dump the first three stages of pdf_dic to a json file
|
||||
if para_debug_mode == "full":
|
||||
pdf_dic_json_fpath = __save_pdf_dic(pdf_dic, output_pdf_path, stage="0", para_debug_mode=para_debug_mode)
|
||||
|
||||
"""
|
||||
Detect titles in the document
|
||||
"""
|
||||
doc_statistics = pdf_dic["statistics"]
|
||||
titleProcessor = TitleProcessor(doc_statistics)
|
||||
pdf_dic = titleProcessor.batch_process_blocks_detect_titles(pdf_dic)
|
||||
|
||||
if para_debug_mode == "full":
|
||||
pdf_dic_json_fpath = __save_pdf_dic(pdf_dic, output_pdf_path, stage="1", para_debug_mode=para_debug_mode)
|
||||
|
||||
"""
|
||||
Detect and divide the level of the titles
|
||||
"""
|
||||
titleProcessor = TitleProcessor()
|
||||
|
||||
pdf_dic = titleProcessor.batch_process_blocks_recog_title_level(pdf_dic)
|
||||
|
||||
if para_debug_mode == "full":
|
||||
pdf_dic_json_fpath = __save_pdf_dic(pdf_dic, output_pdf_path, stage="2", para_debug_mode=para_debug_mode)
|
||||
|
||||
"""
|
||||
Detect and split paragraphs inside each block
|
||||
"""
|
||||
blockInnerParasProcessor = BlockTerminationProcessor()
|
||||
|
||||
pdf_dic = blockInnerParasProcessor.batch_process_blocks(pdf_dic)
|
||||
|
||||
if para_debug_mode == "full":
|
||||
pdf_dic_json_fpath = __save_pdf_dic(pdf_dic, output_pdf_path, stage="3", para_debug_mode=para_debug_mode)
|
||||
|
||||
# pdf_dic_json_fpath = __save_pdf_dic(pdf_dic, output_pdf_path, stage="3", para_debug_mode="full")
|
||||
# print_green(f"pdf_dic_json_fpath: {pdf_dic_json_fpath}")
|
||||
|
||||
"""
|
||||
Detect and combine paragraphs from different blocks into one paragraph
|
||||
"""
|
||||
blockContinuationProcessor = BlockContinuationProcessor()
|
||||
|
||||
pdf_dic = blockContinuationProcessor.batch_tag_paras(pdf_dic)
|
||||
pdf_dic = blockContinuationProcessor.batch_merge_paras(pdf_dic)
|
||||
|
||||
if para_debug_mode == "full":
|
||||
pdf_dic_json_fpath = __save_pdf_dic(pdf_dic, output_pdf_path, stage="4", para_debug_mode=para_debug_mode)
|
||||
|
||||
# pdf_dic_json_fpath = __save_pdf_dic(pdf_dic, output_pdf_path, stage="4", para_debug_mode="full")
|
||||
# print_green(f"pdf_dic_json_fpath: {pdf_dic_json_fpath}")
|
||||
|
||||
"""
|
||||
Discard pdf files by checking exceptions and return the error info to the caller
|
||||
"""
|
||||
discardByException = DiscardByException()
|
||||
|
||||
is_discard_by_single_line_block = discardByException.discard_by_single_line_block(
|
||||
pdf_dic, exception=DenseSingleLineBlockException()
|
||||
)
|
||||
is_discard_by_title_detection = discardByException.discard_by_title_detection(
|
||||
pdf_dic, exception=TitleDetectionException()
|
||||
)
|
||||
is_discard_by_title_level = discardByException.discard_by_title_level(pdf_dic, exception=TitleLevelException())
|
||||
is_discard_by_split_para = discardByException.discard_by_split_para(pdf_dic, exception=ParaSplitException())
|
||||
is_discard_by_merge_para = discardByException.discard_by_merge_para(pdf_dic, exception=ParaMergeException())
|
||||
|
||||
"""
|
||||
if any(
|
||||
info is not None
|
||||
for info in [
|
||||
is_discard_by_single_line_block,
|
||||
is_discard_by_title_detection,
|
||||
is_discard_by_title_level,
|
||||
is_discard_by_split_para,
|
||||
is_discard_by_merge_para,
|
||||
]
|
||||
):
|
||||
error_info = next(
|
||||
(
|
||||
info
|
||||
for info in [
|
||||
is_discard_by_single_line_block,
|
||||
is_discard_by_title_detection,
|
||||
is_discard_by_title_level,
|
||||
is_discard_by_split_para,
|
||||
is_discard_by_merge_para,
|
||||
]
|
||||
if info is not None
|
||||
),
|
||||
None,
|
||||
)
|
||||
return pdf_dic, error_info
|
||||
|
||||
if any(
|
||||
info is not None
|
||||
for info in [
|
||||
is_discard_by_single_line_block,
|
||||
is_discard_by_title_detection,
|
||||
is_discard_by_title_level,
|
||||
is_discard_by_split_para,
|
||||
is_discard_by_merge_para,
|
||||
]
|
||||
):
|
||||
error_info = next(
|
||||
(
|
||||
info
|
||||
for info in [
|
||||
is_discard_by_single_line_block,
|
||||
is_discard_by_title_detection,
|
||||
is_discard_by_title_level,
|
||||
is_discard_by_split_para,
|
||||
is_discard_by_merge_para,
|
||||
]
|
||||
if info is not None
|
||||
),
|
||||
None,
|
||||
)
|
||||
return pdf_dic, error_info
|
||||
"""
|
||||
|
||||
"""
|
||||
Dump the final pdf_dic to a json file
|
||||
"""
|
||||
if para_debug_mode is not None:
|
||||
with open(output_json_file, "w", encoding="utf-8") as f:
|
||||
json.dump(pdf_info_dict, f, ensure_ascii=False, indent=4)
|
||||
|
||||
"""
|
||||
Draw the annotations
|
||||
"""
|
||||
|
||||
if is_discard_by_single_line_block is not None:
|
||||
error_info = is_discard_by_single_line_block
|
||||
elif is_discard_by_title_detection is not None:
|
||||
error_info = is_discard_by_title_detection
|
||||
elif is_discard_by_title_level is not None:
|
||||
error_info = is_discard_by_title_level
|
||||
elif is_discard_by_split_para is not None:
|
||||
error_info = is_discard_by_split_para
|
||||
elif is_discard_by_merge_para is not None:
|
||||
error_info = is_discard_by_merge_para
|
||||
|
||||
if error_info is not None:
|
||||
return pdf_dic, error_info
|
||||
|
||||
"""
|
||||
Dump the final pdf_dic to a json file
|
||||
"""
|
||||
if para_debug_mode is not None:
|
||||
with open(output_json_file, "w", encoding="utf-8") as f:
|
||||
json.dump(pdf_info_dict, f, ensure_ascii=False, indent=4)
|
||||
|
||||
"""
|
||||
Draw the annotations
|
||||
"""
|
||||
if para_debug_mode is not None:
|
||||
drawAnnos = DrawAnnos()
|
||||
drawAnnos.draw_annos(input_pdf_path, pdf_dic, output_pdf_path)
|
||||
|
||||
"""
|
||||
Remove the intermediate files which are generated in the process of paragraph processing if debug_mode is simple
|
||||
"""
|
||||
if para_debug_mode is not None:
|
||||
for fpath in os.listdir(output_dir):
|
||||
if fpath.endswith(".json") and "stage" in fpath:
|
||||
os.remove(os.path.join(output_dir, fpath))
|
||||
|
||||
return pdf_dic, error_info
|
||||
@@ -0,0 +1,209 @@
|
||||
from para.commons import *
|
||||
|
||||
class RawBlockProcessor:
|
||||
def __init__(self) -> None:
|
||||
self.y_tolerance = 2
|
||||
self.pdf_dic = {}
|
||||
|
||||
def __span_flags_decomposer(self, span_flags):
|
||||
"""
|
||||
Make font flags human readable.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
self : object
|
||||
The instance of the class.
|
||||
|
||||
span_flags : int
|
||||
span flags
|
||||
|
||||
Returns
|
||||
-------
|
||||
l : dict
|
||||
decomposed flags
|
||||
"""
|
||||
|
||||
l = {
|
||||
"is_superscript": False,
|
||||
"is_italic": False,
|
||||
"is_serifed": False,
|
||||
"is_sans_serifed": False,
|
||||
"is_monospaced": False,
|
||||
"is_proportional": False,
|
||||
"is_bold": False,
|
||||
}
|
||||
|
||||
if span_flags & 2**0:
|
||||
l["is_superscript"] = True # 表示上标
|
||||
|
||||
if span_flags & 2**1:
|
||||
l["is_italic"] = True # 表示斜体
|
||||
|
||||
if span_flags & 2**2:
|
||||
l["is_serifed"] = True # 表示衬线字体
|
||||
else:
|
||||
l["is_sans_serifed"] = True # 表示非衬线字体
|
||||
|
||||
if span_flags & 2**3:
|
||||
l["is_monospaced"] = True # 表示等宽字体
|
||||
else:
|
||||
l["is_proportional"] = True # 表示比例字体
|
||||
|
||||
if span_flags & 2**4:
|
||||
l["is_bold"] = True # 表示粗体
|
||||
|
||||
return l
|
||||
|
||||
def __make_new_lines(self, raw_lines):
|
||||
"""
|
||||
This function makes new lines.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
self : object
|
||||
The instance of the class.
|
||||
|
||||
raw_lines : list
|
||||
raw lines
|
||||
|
||||
Returns
|
||||
-------
|
||||
new_lines : list
|
||||
new lines
|
||||
"""
|
||||
new_lines = []
|
||||
new_line = None
|
||||
|
||||
for raw_line in raw_lines:
|
||||
raw_line_bbox = raw_line["bbox"]
|
||||
raw_line_spans = raw_line["spans"]
|
||||
raw_line_text = "".join([span["text"] for span in raw_line_spans])
|
||||
raw_line_dir = raw_line.get("dir", None)
|
||||
|
||||
decomposed_line_spans = []
|
||||
for span in raw_line_spans:
|
||||
raw_flags = span["flags"]
|
||||
decomposed_flags = self.__span_flags_decomposer(raw_flags)
|
||||
span["decomposed_flags"] = decomposed_flags
|
||||
decomposed_line_spans.append(span)
|
||||
|
||||
if new_line is None:
|
||||
new_line = {
|
||||
"bbox": raw_line_bbox,
|
||||
"text": raw_line_text,
|
||||
"dir": raw_line_dir if raw_line_dir else (0, 0),
|
||||
"spans": decomposed_line_spans,
|
||||
}
|
||||
else:
|
||||
if (
|
||||
abs(raw_line_bbox[1] - new_line["bbox"][1]) <= self.y_tolerance
|
||||
and abs(raw_line_bbox[3] - new_line["bbox"][3]) <= self.y_tolerance
|
||||
):
|
||||
new_line["bbox"] = (
|
||||
min(new_line["bbox"][0], raw_line_bbox[0]), # left
|
||||
new_line["bbox"][1], # top
|
||||
max(new_line["bbox"][2], raw_line_bbox[2]), # right
|
||||
raw_line_bbox[3], # bottom
|
||||
)
|
||||
new_line["text"] += " " + raw_line_text
|
||||
new_line["spans"].extend(raw_line_spans)
|
||||
new_line["dir"] = (
|
||||
new_line["dir"][0] + raw_line_dir[0],
|
||||
new_line["dir"][1] + raw_line_dir[1],
|
||||
)
|
||||
else:
|
||||
new_lines.append(new_line)
|
||||
new_line = {
|
||||
"bbox": raw_line_bbox,
|
||||
"text": raw_line_text,
|
||||
"dir": raw_line_dir if raw_line_dir else (0, 0),
|
||||
"spans": raw_line_spans,
|
||||
}
|
||||
if new_line:
|
||||
new_lines.append(new_line)
|
||||
|
||||
return new_lines
|
||||
|
||||
def __make_new_block(self, raw_block):
|
||||
"""
|
||||
This function makes a new block.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
self : object
|
||||
The instance of the class.
|
||||
----------
|
||||
raw_block : dict
|
||||
a raw block
|
||||
|
||||
Returns
|
||||
-------
|
||||
new_block : dict
|
||||
|
||||
Schema of new_block:
|
||||
{
|
||||
"block_id": "block_1",
|
||||
"bbox": [0, 0, 100, 100],
|
||||
"text": "This is a block.",
|
||||
"lines": [
|
||||
{
|
||||
"bbox": [0, 0, 100, 100],
|
||||
"text": "This is a line.",
|
||||
"spans": [
|
||||
{
|
||||
"text": "This is a span.",
|
||||
"font": "Times New Roman",
|
||||
"size": 12,
|
||||
"color": "#000000",
|
||||
}
|
||||
],
|
||||
}
|
||||
],
|
||||
}
|
||||
"""
|
||||
new_block = {}
|
||||
|
||||
block_id = raw_block["number"]
|
||||
block_bbox = raw_block["bbox"]
|
||||
block_text = " ".join(span["text"] for line in raw_block["lines"] for span in line["spans"])
|
||||
raw_lines = raw_block["lines"]
|
||||
block_lines = self.__make_new_lines(raw_lines)
|
||||
|
||||
new_block["block_id"] = block_id
|
||||
new_block["bbox"] = block_bbox
|
||||
new_block["text"] = block_text
|
||||
new_block["lines"] = block_lines
|
||||
|
||||
return new_block
|
||||
|
||||
def batch_process_blocks(self, pdf_dic):
|
||||
"""
|
||||
This function processes the blocks in batch.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
self : object
|
||||
The instance of the class.
|
||||
----------
|
||||
blocks : list
|
||||
Input block is a list of raw blocks. Schema can refer to the value of key ""preproc_blocks", demo file is app/pdf_toolbox/test/preproc_2_parasplit_example.json.
|
||||
|
||||
Returns
|
||||
-------
|
||||
result_dict : dict
|
||||
result dictionary
|
||||
"""
|
||||
|
||||
for page_id, blocks in pdf_dic.items():
|
||||
if page_id.startswith("page_"):
|
||||
para_blocks = []
|
||||
if "preproc_blocks" in blocks.keys():
|
||||
input_blocks = blocks["preproc_blocks"]
|
||||
for raw_block in input_blocks:
|
||||
new_block = self.__make_new_block(raw_block)
|
||||
para_blocks.append(new_block)
|
||||
|
||||
blocks["para_blocks"] = para_blocks
|
||||
|
||||
return pdf_dic
|
||||
|
||||
+269
@@ -0,0 +1,269 @@
|
||||
import sys
|
||||
from collections import Counter
|
||||
import numpy as np
|
||||
|
||||
from para.commons import *
|
||||
|
||||
|
||||
if sys.version_info[0] >= 3:
|
||||
sys.stdout.reconfigure(encoding="utf-8") # type: ignore
|
||||
|
||||
|
||||
class BlockStatisticsCalculator:
|
||||
def __init__(self) -> None:
|
||||
pass
|
||||
|
||||
def __calc_stats_of_new_lines(self, new_lines):
|
||||
"""
|
||||
This function calculates the paragraph metrics
|
||||
|
||||
Parameters
|
||||
----------
|
||||
combined_lines : list
|
||||
combined lines
|
||||
|
||||
Returns
|
||||
-------
|
||||
X0 : float
|
||||
Median of x0 values, which represents the left average boundary of the block
|
||||
X1 : float
|
||||
Median of x1 values, which represents the right average boundary of the block
|
||||
avg_char_width : float
|
||||
Average of char widths, which represents the average char width of the block
|
||||
avg_char_height : float
|
||||
Average of line heights, which represents the average line height of the block
|
||||
|
||||
"""
|
||||
x0_values = []
|
||||
x1_values = []
|
||||
char_widths = []
|
||||
char_heights = []
|
||||
|
||||
block_font_types = []
|
||||
block_font_sizes = []
|
||||
block_directions = []
|
||||
|
||||
if len(new_lines) > 0:
|
||||
for i, line in enumerate(new_lines):
|
||||
line_bbox = line["bbox"]
|
||||
line_text = line["text"]
|
||||
line_spans = line["spans"]
|
||||
|
||||
num_chars = len([ch for ch in line_text if not ch.isspace()])
|
||||
|
||||
x0_values.append(line_bbox[0])
|
||||
x1_values.append(line_bbox[2])
|
||||
|
||||
if num_chars > 0:
|
||||
char_width = (line_bbox[2] - line_bbox[0]) / num_chars
|
||||
char_widths.append(char_width)
|
||||
|
||||
for span in line_spans:
|
||||
block_font_types.append(span["font"])
|
||||
block_font_sizes.append(span["size"])
|
||||
|
||||
if "dir" in line:
|
||||
block_directions.append(line["dir"])
|
||||
|
||||
# line_font_types = [span["font"] for span in line_spans]
|
||||
char_heights = [span["size"] for span in line_spans]
|
||||
|
||||
X0 = np.median(x0_values) if x0_values else 0
|
||||
X1 = np.median(x1_values) if x1_values else 0
|
||||
avg_char_width = sum(char_widths) / len(char_widths) if char_widths else 0
|
||||
avg_char_height = sum(char_heights) / len(char_heights) if char_heights else 0
|
||||
|
||||
# max_freq_font_type = max(set(block_font_types), key=block_font_types.count) if block_font_types else None
|
||||
|
||||
max_span_length = 0
|
||||
max_span_font_type = None
|
||||
for line in new_lines:
|
||||
line_spans = line["spans"]
|
||||
for span in line_spans:
|
||||
span_length = span["bbox"][2] - span["bbox"][0]
|
||||
if span_length > max_span_length:
|
||||
max_span_length = span_length
|
||||
max_span_font_type = span["font"]
|
||||
|
||||
max_freq_font_type = max_span_font_type
|
||||
|
||||
avg_font_size = sum(block_font_sizes) / len(block_font_sizes) if block_font_sizes else None
|
||||
|
||||
avg_dir_horizontal = sum([dir[0] for dir in block_directions]) / len(block_directions) if block_directions else 0
|
||||
avg_dir_vertical = sum([dir[1] for dir in block_directions]) / len(block_directions) if block_directions else 0
|
||||
|
||||
median_font_size = float(np.median(block_font_sizes)) if block_font_sizes else None
|
||||
|
||||
return (
|
||||
X0,
|
||||
X1,
|
||||
avg_char_width,
|
||||
avg_char_height,
|
||||
max_freq_font_type,
|
||||
avg_font_size,
|
||||
(avg_dir_horizontal, avg_dir_vertical),
|
||||
median_font_size,
|
||||
)
|
||||
|
||||
def __make_new_block(self, input_block):
|
||||
new_block = {}
|
||||
|
||||
raw_lines = input_block["lines"]
|
||||
stats = self.__calc_stats_of_new_lines(raw_lines)
|
||||
|
||||
block_id = input_block["block_id"]
|
||||
block_bbox = input_block["bbox"]
|
||||
block_text = input_block["text"]
|
||||
block_lines = raw_lines
|
||||
block_avg_left_boundary = stats[0]
|
||||
block_avg_right_boundary = stats[1]
|
||||
block_avg_char_width = stats[2]
|
||||
block_avg_char_height = stats[3]
|
||||
block_font_type = stats[4]
|
||||
block_font_size = stats[5]
|
||||
block_direction = stats[6]
|
||||
block_median_font_size = stats[7]
|
||||
|
||||
new_block["block_id"] = block_id
|
||||
new_block["bbox"] = block_bbox
|
||||
new_block["text"] = block_text
|
||||
new_block["dir"] = block_direction
|
||||
new_block["X0"] = block_avg_left_boundary
|
||||
new_block["X1"] = block_avg_right_boundary
|
||||
new_block["avg_char_width"] = block_avg_char_width
|
||||
new_block["avg_char_height"] = block_avg_char_height
|
||||
new_block["block_font_type"] = block_font_type
|
||||
new_block["block_font_size"] = block_font_size
|
||||
new_block["lines"] = block_lines
|
||||
new_block["median_font_size"] = block_median_font_size
|
||||
|
||||
return new_block
|
||||
|
||||
def batch_process_blocks(self, pdf_dic):
|
||||
"""
|
||||
This function processes the blocks in batch.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
self : object
|
||||
The instance of the class.
|
||||
----------
|
||||
blocks : list
|
||||
Input block is a list of raw blocks. Schema can refer to the value of key ""preproc_blocks", demo file is app/pdf_toolbox/test/preproc_2_parasplit_example.json
|
||||
|
||||
Returns
|
||||
-------
|
||||
result_dict : dict
|
||||
result dictionary
|
||||
"""
|
||||
|
||||
for page_id, blocks in pdf_dic.items():
|
||||
if page_id.startswith("page_"):
|
||||
para_blocks = []
|
||||
if "para_blocks" in blocks.keys():
|
||||
input_blocks = blocks["para_blocks"]
|
||||
for input_block in input_blocks:
|
||||
new_block = self.__make_new_block(input_block)
|
||||
para_blocks.append(new_block)
|
||||
|
||||
blocks["para_blocks"] = para_blocks
|
||||
|
||||
return pdf_dic
|
||||
|
||||
|
||||
class DocStatisticsCalculator:
|
||||
def __init__(self) -> None:
|
||||
pass
|
||||
|
||||
def calc_stats_of_doc(self, pdf_dict):
|
||||
"""
|
||||
This function computes the statistics of the document
|
||||
|
||||
Parameters
|
||||
----------
|
||||
result_dict : dict
|
||||
result dictionary
|
||||
|
||||
Returns
|
||||
-------
|
||||
statistics : dict
|
||||
statistics of the document
|
||||
"""
|
||||
|
||||
total_text_length = 0
|
||||
total_num_blocks = 0
|
||||
|
||||
for page_id, blocks in pdf_dict.items():
|
||||
if page_id.startswith("page_"):
|
||||
if "para_blocks" in blocks.keys():
|
||||
para_blocks = blocks["para_blocks"]
|
||||
for para_block in para_blocks:
|
||||
total_text_length += len(para_block["text"])
|
||||
total_num_blocks += 1
|
||||
|
||||
avg_text_length = total_text_length / total_num_blocks if total_num_blocks else 0
|
||||
|
||||
font_list = []
|
||||
|
||||
for page_id, blocks in pdf_dict.items():
|
||||
if page_id.startswith("page_"):
|
||||
if "para_blocks" in blocks.keys():
|
||||
input_blocks = blocks["para_blocks"]
|
||||
for input_block in input_blocks:
|
||||
block_text_length = len(input_block.get("text", ""))
|
||||
if block_text_length < avg_text_length * 0.5:
|
||||
continue
|
||||
block_font_type = safe_get(input_block, "block_font_type", "")
|
||||
block_font_size = safe_get(input_block, "block_font_size", 0)
|
||||
font_list.append((block_font_type, block_font_size))
|
||||
|
||||
font_counter = Counter(font_list)
|
||||
most_common_font = font_counter.most_common(1)[0] if font_list else (("", 0), 0)
|
||||
second_most_common_font = font_counter.most_common(2)[1] if len(font_counter) > 1 else (("", 0), 0)
|
||||
|
||||
statistics = {
|
||||
"num_pages": 0,
|
||||
"num_blocks": 0,
|
||||
"num_paras": 0,
|
||||
"num_titles": 0,
|
||||
"num_header_blocks": 0,
|
||||
"num_footer_blocks": 0,
|
||||
"num_watermark_blocks": 0,
|
||||
"num_vertical_margin_note_blocks": 0,
|
||||
"most_common_font_type": most_common_font[0][0],
|
||||
"most_common_font_size": most_common_font[0][1],
|
||||
"number_of_most_common_font": most_common_font[1],
|
||||
"second_most_common_font_type": second_most_common_font[0][0],
|
||||
"second_most_common_font_size": second_most_common_font[0][1],
|
||||
"number_of_second_most_common_font": second_most_common_font[1],
|
||||
"avg_text_length": avg_text_length,
|
||||
}
|
||||
|
||||
for page_id, blocks in pdf_dict.items():
|
||||
if page_id.startswith("page_"):
|
||||
blocks = pdf_dict[page_id]["para_blocks"]
|
||||
statistics["num_pages"] += 1
|
||||
for block_id, block_data in enumerate(blocks):
|
||||
statistics["num_blocks"] += 1
|
||||
|
||||
if "paras" in block_data.keys():
|
||||
statistics["num_paras"] += len(block_data["paras"])
|
||||
|
||||
for line in block_data["lines"]:
|
||||
if line.get("is_title", 0):
|
||||
statistics["num_titles"] += 1
|
||||
|
||||
if block_data.get("is_header", 0):
|
||||
statistics["num_header_blocks"] += 1
|
||||
if block_data.get("is_footer", 0):
|
||||
statistics["num_footer_blocks"] += 1
|
||||
if block_data.get("is_watermark", 0):
|
||||
statistics["num_watermark_blocks"] += 1
|
||||
if block_data.get("is_vertical_margin_note", 0):
|
||||
statistics["num_vertical_margin_note_blocks"] += 1
|
||||
|
||||
pdf_dict["statistics"] = statistics
|
||||
|
||||
return pdf_dict
|
||||
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,317 @@
|
||||
import sys
|
||||
from typing import Tuple
|
||||
import os
|
||||
import click
|
||||
import boto3, json
|
||||
from botocore.config import Config
|
||||
from libs.commons import fitz
|
||||
from loguru import logger
|
||||
from pathlib import Path
|
||||
from tqdm import tqdm
|
||||
import numpy as np
|
||||
|
||||
# sys.path.insert(0, "/mnt/petrelfs/ouyanglinke/code-clean/")
|
||||
# print(sys.path)
|
||||
|
||||
from validation import cal_edit_distance, format_gt_bbox, label_match, detect_val
|
||||
|
||||
|
||||
# from pdf2text_recogFigure_20231107 import parse_images # 获取figures的bbox
|
||||
# from pdf2text_recogTable_20231107 import parse_tables # 获取tables的bbox
|
||||
# from pdf2text_recogEquation_20231108 import parse_equations # 获取equations的bbox
|
||||
# from pdf2text_recogTitle_20231113 import parse_titles # 获取Title的bbox
|
||||
# from pdf2text_recogPara import parse_blocks_per_page
|
||||
# from bbox_sort import bbox_sort, CONTENT_IDX, CONTENT_TYPE_IDX
|
||||
|
||||
from layout.bbox_sort import bbox_sort, CONTENT_IDX, CONTENT_TYPE_IDX
|
||||
from pdf2text_recogFigure import parse_images # 获取figures的bbox
|
||||
from pdf2text_recogTable import parse_tables # 获取tables的bbox
|
||||
from pdf2text_recogEquation import parse_equations # 获取equations的bbox
|
||||
from pdf2text_recogTitle import parse_titles # 获取titles的bbox
|
||||
from pdf2text_recogHeader import parse_headers # 获取headers的bbox
|
||||
from pdf2text_recogPageNo import parse_pageNos # 获取pageNos的bbox
|
||||
# from pdf2text_recogFootnote import parse_footnotes # 获取footnotes的bbox
|
||||
from pdf2text_recogFooter import parse_footers # 获取footers的bbox
|
||||
from pdf2text_evaluatePdfLayout import evaluate_pdf_layout # 评估页面的Layout是否是规整的。
|
||||
from pdf2text_recogPara import process_blocks_per_page, postprocess_paras_pipeline
|
||||
from libs.commons import parse_aws_param, parse_bucket_key, read_file, join_path
|
||||
|
||||
|
||||
def cut_image(bbox: Tuple, page_num: int, page: fitz.Page, save_parent_path: str, s3_profile: str):
|
||||
"""
|
||||
从第page_num页的page中,根据bbox进行裁剪出一张jpg图片,返回图片路径
|
||||
save_path:需要同时支持s3和本地, 图片存放在save_path下,文件名是: {page_num}_{bbox[0]}_{bbox[1]}_{bbox[2]}_{bbox[3]}.jpg , bbox内数字取整。
|
||||
"""
|
||||
# 拼接路径
|
||||
image_save_path = join_path(save_parent_path, f"{page_num}_{int(bbox[0])}_{int(bbox[1])}_{int(bbox[2])}_{int(bbox[3])}.jpg")
|
||||
try:
|
||||
# 将坐标转换为fitz.Rect对象
|
||||
rect = fitz.Rect(*bbox)
|
||||
# 配置缩放倍数为3倍
|
||||
zoom = fitz.Matrix(3, 3)
|
||||
# 截取图片
|
||||
pix = page.get_pixmap(clip=rect, matrix=zoom)
|
||||
|
||||
# 打印图片文件名
|
||||
# print(f"Saved {image_save_path}")
|
||||
if image_save_path.startswith("s3://"):
|
||||
ak, sk, end_point, addressing_style = parse_aws_param(s3_profile)
|
||||
cli = boto3.client(service_name="s3", aws_access_key_id=ak, aws_secret_access_key=sk, endpoint_url=end_point,
|
||||
config=Config(s3={'addressing_style': addressing_style}))
|
||||
bucket_name, bucket_key = parse_bucket_key(image_save_path)
|
||||
# 将字节流上传到s3
|
||||
cli.upload_fileobj(pix.tobytes(output='jpeg', jpg_quality=95), bucket_name, bucket_key)
|
||||
else:
|
||||
# 保存图片到本地
|
||||
# 先检查一下image_save_path的父目录是否存在,如果不存在,就创建
|
||||
parent_dir = os.path.dirname(image_save_path)
|
||||
if not os.path.exists(parent_dir):
|
||||
os.makedirs(parent_dir)
|
||||
pix.save(image_save_path, jpg_quality=95)
|
||||
# 为了直接能在markdown里看,这里把地址改为相对于mardown的地址
|
||||
pth = Path(image_save_path)
|
||||
image_save_path = f"{pth.parent.name}/{pth.name}"
|
||||
return image_save_path
|
||||
except Exception as e:
|
||||
logger.exception(e)
|
||||
return image_save_path
|
||||
|
||||
|
||||
|
||||
def get_images_by_bboxes(book_name:str, page_num:int, page: fitz.Page, save_path:str, s3_profile:str, image_bboxes:list, table_bboxes:list, equation_inline_bboxes:list, equation_interline_bboxes:list) -> dict:
|
||||
"""
|
||||
返回一个dict, key为bbox, 值是图片地址
|
||||
"""
|
||||
ret = {}
|
||||
|
||||
# 图片的保存路径组成是这样的: {s3_or_local_path}/{book_name}/{images|tables|equations}/{page_num}_{bbox[0]}_{bbox[1]}_{bbox[2]}_{bbox[3]}.jpg
|
||||
image_save_path = join_path(save_path, book_name, "images")
|
||||
table_save_path = join_path(save_path, book_name, "tables")
|
||||
equation_inline_save_path = join_path(save_path, book_name, "equations_inline")
|
||||
equation_interline_save_path = join_path(save_path, book_name, "equation_interline")
|
||||
|
||||
for bbox in image_bboxes:
|
||||
image_path = cut_image(bbox, page_num, page, image_save_path, s3_profile)
|
||||
ret[bbox] = (image_path, "image") # 第二个元素是"image",表示是图片
|
||||
|
||||
for bbox in table_bboxes:
|
||||
image_path = cut_image(bbox, page_num, page, table_save_path, s3_profile)
|
||||
ret[bbox] = (image_path, "table")
|
||||
|
||||
# 对公式目前只截图,不返回
|
||||
for bbox in equation_inline_bboxes:
|
||||
cut_image(bbox, page_num, page, equation_inline_save_path, s3_profile)
|
||||
|
||||
for bbox in equation_interline_bboxes:
|
||||
cut_image(bbox, page_num, page, equation_interline_save_path, s3_profile)
|
||||
|
||||
return ret
|
||||
|
||||
def reformat_bboxes(images_box_path_dict:list, paras_dict:dict):
|
||||
"""
|
||||
把bbox重新组装成一个list,每个元素[x0, y0, x1, y1, block_content, idx_x, idx_y], 初始时候idx_x, idx_y都是None. 对于图片、公式来说,block_content是图片的地址, 对于段落来说,block_content是段落的内容
|
||||
"""
|
||||
all_bboxes = []
|
||||
for bbox, image_info in images_box_path_dict.items():
|
||||
all_bboxes.append([bbox[0], bbox[1], bbox[2], bbox[3], image_info, None, None, 'image'])
|
||||
|
||||
paras_dict = paras_dict[f"page_{paras_dict['page_id']}"]
|
||||
|
||||
for block_id, kvpair in paras_dict.items():
|
||||
bbox = kvpair['bbox']
|
||||
content = kvpair
|
||||
all_bboxes.append([bbox[0], bbox[1], bbox[2], bbox[3], content, None, None, 'text'])
|
||||
|
||||
return all_bboxes
|
||||
|
||||
|
||||
def concat2markdown(all_bboxes:list):
|
||||
"""
|
||||
对排序后的bboxes拼接内容
|
||||
"""
|
||||
content_md = ""
|
||||
for box in all_bboxes:
|
||||
content_type = box[CONTENT_TYPE_IDX]
|
||||
if content_type == 'image':
|
||||
image_type = box[CONTENT_IDX][1]
|
||||
image_path = box[CONTENT_IDX][0]
|
||||
content_md += f""
|
||||
content_md += "\n\n"
|
||||
elif content_type == 'text': # 组装文本
|
||||
paras = box[CONTENT_IDX]['paras']
|
||||
text_content = ""
|
||||
for para_id, para in paras.items():# 拼装内部的段落文本
|
||||
text_content += para['text']
|
||||
text_content += "\n\n"
|
||||
|
||||
content_md += text_content
|
||||
else:
|
||||
raise Exception(f"ERROR: {content_type} is not supported!")
|
||||
|
||||
return content_md
|
||||
|
||||
|
||||
|
||||
def main(s3_pdf_path: str, s3_pdf_profile: str, pdf_model_path:str, pdf_model_profile:str, save_path: str, page_num: int):
|
||||
"""
|
||||
|
||||
"""
|
||||
pth = Path(s3_pdf_path)
|
||||
book_name = pth.name
|
||||
#book_name = "".join(os.path.basename(s3_pdf_path).split(".")[0:-1])
|
||||
res_dir_path = None
|
||||
exclude_bboxes = []
|
||||
# text_content_save_path = f"{save_path}/{book_name}/book.md"
|
||||
# metadata_save_path = f"{save_path}/{book_name}/metadata.json"
|
||||
|
||||
try:
|
||||
pdf_bytes = read_file(s3_pdf_path, s3_pdf_profile)
|
||||
pdf_docs = fitz.open("pdf", pdf_bytes)
|
||||
page_id = page_num - 1
|
||||
page = pdf_docs[page_id] # 验证集只需要读取特定页面即可
|
||||
model_output_json = join_path(pdf_model_path, f"page_{page_num}.json") # 模型输出的页面编号从1开始的
|
||||
json_from_docx = read_file(model_output_json, pdf_model_profile) # TODO 这个读取方法名字应该改一下,避免语义歧义
|
||||
json_from_docx_obj = json.loads(json_from_docx)
|
||||
|
||||
|
||||
# 解析图片
|
||||
image_bboxes = parse_images(page_id, page, json_from_docx_obj)
|
||||
|
||||
# 解析表格
|
||||
table_bboxes = parse_tables(page_id, page, json_from_docx_obj)
|
||||
|
||||
# 解析公式
|
||||
equations_interline_bboxes, equations_inline_bboxes = parse_equations(page_id, page, json_from_docx_obj)
|
||||
|
||||
# # 解析标题
|
||||
# title_bboxs = parse_titles(page_id, page, res_dir_path, json_from_docx_obj, exclude_bboxes)
|
||||
# # 解析页眉
|
||||
# header_bboxs = parse_headers(page_id, page, res_dir_path, json_from_docx_obj, exclude_bboxes)
|
||||
# # 解析页码
|
||||
# pageNo_bboxs = parse_pageNos(page_id, page, res_dir_path, json_from_docx_obj, exclude_bboxes)
|
||||
# # 解析脚注
|
||||
# footnote_bboxs = parse_footnotes(page_id, page, res_dir_path, json_from_docx_obj, exclude_bboxes)
|
||||
# # 解析页脚
|
||||
# footer_bboxs = parse_footers(page_id, page, res_dir_path, json_from_docx_obj, exclude_bboxes)
|
||||
# # 评估Layout是否规整、简单
|
||||
# isSimpleLayout_flag, fullColumn_cnt, subColumn_cnt, curPage_loss = evaluate_pdf_layout(page_id, page, res_dir_path, json_from_docx_obj, exclude_bboxes)
|
||||
|
||||
# 把图、表、公式都进行截图,保存到本地,返回图片路径作为内容
|
||||
images_box_path_dict = get_images_by_bboxes(book_name, page_id, page, save_path, s3_pdf_profile, image_bboxes, table_bboxes, equations_inline_bboxes,
|
||||
equations_interline_bboxes) # 只要表格和图片的截图
|
||||
|
||||
# 解析文字段落
|
||||
|
||||
footer_bboxes = []
|
||||
header_bboxes = []
|
||||
exclude_bboxes = image_bboxes + table_bboxes
|
||||
paras_dict = process_blocks_per_page(page, page_id, image_bboxes, table_bboxes, equations_inline_bboxes, equations_interline_bboxes, footer_bboxes, header_bboxes)
|
||||
# paras_dict = postprocess_paras_pipeline(paras_dict)
|
||||
|
||||
# 最后一步,根据bbox进行从左到右,从上到下的排序,之后拼接起来, 排序
|
||||
|
||||
all_bboxes = reformat_bboxes(images_box_path_dict, paras_dict) # 由于公式目前还没有,所以equation_bboxes是None,多数存在段落里,暂时不解析
|
||||
|
||||
|
||||
# 返回的是一个数组,每个元素[x0, y0, x1, y1, block_content, idx_x, idx_y, type], 初始时候idx_x, idx_y都是None. 对于图片、公式来说,block_content是图片的地址, 对于段落来说,block_content是段落的内容
|
||||
# sorted_bboxes = bbox_sort(all_bboxes)
|
||||
|
||||
# markdown_text = concat2markdown(sorted_bboxes)
|
||||
|
||||
# parent_dir = os.path.dirname(text_content_save_path)
|
||||
# if not os.path.exists(parent_dir):
|
||||
# os.makedirs(parent_dir)
|
||||
|
||||
# with open(text_content_save_path, "a") as f:
|
||||
# f.write(markdown_text)
|
||||
# f.write(chr(12)) #换页符
|
||||
# end for
|
||||
# 写一个小的json,记录元数据
|
||||
# metadata = {"book_name": book_name, "pdf_path": s3_pdf_path, "pdf_model_path": pdf_model_path, "save_path": save_path}
|
||||
# with open(metadata_save_path, "w") as f:
|
||||
# json.dump(metadata, f, ensure_ascii=False, indent=4)
|
||||
|
||||
return all_bboxes
|
||||
|
||||
except Exception as e:
|
||||
print(f"ERROR: {s3_pdf_path}, {e}", file=sys.stderr)
|
||||
logger.exception(e)
|
||||
|
||||
|
||||
# @click.command()
|
||||
# @click.option('--pdf-file-sub-path', help='s3上pdf文件的路径')
|
||||
# @click.option('--save-path', help='解析出来的图片,文本的保存父目录')
|
||||
def validation(validation_dataset: str, pdf_bin_file_profile: str, pdf_model_dir: str, pdf_model_profile: str, save_path: str):
|
||||
#pdf_bin_file_path = "s3://llm-raw-snew/llm-raw-scihub/scimag07865000-07865999/10.1007/"
|
||||
# pdf_bin_file_parent_path = "s3://llm-raw-snew/llm-raw-scihub/"
|
||||
|
||||
# pdf_model_parent_dir = "s3://llm-pdf-text/layout_det/scihub/"
|
||||
|
||||
|
||||
# p = Path(pdf_file_sub_path)
|
||||
# pdf_parent_path = p.parent
|
||||
# pdf_file_name = p.name # pdf文件名字,含后缀
|
||||
# pdf_bin_file_path = join_path(pdf_bin_file_parent_path, pdf_parent_path)
|
||||
|
||||
with open(validation_dataset, 'r') as f:
|
||||
samples = json.load(f)
|
||||
|
||||
labels = []
|
||||
det_res = []
|
||||
edit_distance_list = []
|
||||
for sample in tqdm(samples):
|
||||
pdf_name = sample['pdf_name']
|
||||
s3_pdf_path = sample['s3_path']
|
||||
page_num = sample['page']
|
||||
gt_order = sample['order']
|
||||
pre = main(s3_pdf_path, pdf_bin_file_profile, join_path(pdf_model_dir, pdf_name), pdf_model_profile, save_path, page_num)
|
||||
pre_dict_list = []
|
||||
for item in pre:
|
||||
pre_sample = {
|
||||
'box': [item[0],item[1],item[2],item[3]],
|
||||
'type': item[7],
|
||||
'score': 1
|
||||
}
|
||||
pre_dict_list.append(pre_sample)
|
||||
|
||||
det_res.append(pre_dict_list)
|
||||
|
||||
match_change_dict = { # 待确认
|
||||
"figure": "image",
|
||||
"svg_figure": "image",
|
||||
"inline_fomula": "equations_inline",
|
||||
"fomula": "equation_interline",
|
||||
"figure_caption": "text",
|
||||
"table_caption": "text",
|
||||
"fomula_caption": "text"
|
||||
}
|
||||
|
||||
gt_annos = sample['annotations']
|
||||
matched_label = label_match(gt_annos, match_change_dict)
|
||||
labels.append(matched_label)
|
||||
|
||||
# 判断排序函数的精度
|
||||
# 目前不考虑caption与图表相同序号的问题
|
||||
ignore_category = ['abandon', 'figure_caption', 'table_caption', 'formula_caption']
|
||||
gt_bboxes = format_gt_bbox(gt_annos, ignore_category)
|
||||
sorted_bboxes = bbox_sort(gt_bboxes)
|
||||
edit_distance = cal_edit_distance(sorted_bboxes)
|
||||
edit_distance_list.append(edit_distance)
|
||||
|
||||
label_classes = ["image", "text", "table", "equation_interline"]
|
||||
detect_matrix = detect_val(labels, det_res, label_classes)
|
||||
print('detect_matrix', detect_matrix)
|
||||
edit_distance_mean = np.mean(edit_distance_list)
|
||||
print('edit_distance_mean', edit_distance_mean)
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
# 输入可以用以下命令生成批量pdf
|
||||
# aws s3 ls s3://llm-pdf-text/layout_det/scihub/ --profile langchao | tail -n 10 | awk '{print "s3://llm-pdf-text/layout_det/scihub/"$4}' | xargs -I{} aws s3 ls {} --recursive --profile langchao | awk '{print substr($4,19)}' | parallel -j 1 echo {//} | sort -u
|
||||
pdf_bin_file_profile = "outsider"
|
||||
pdf_model_dir = "s3://llm-pdf-text/eval_1k/layout_res/"
|
||||
pdf_model_profile = "langchao"
|
||||
# validation_dataset = "/mnt/petrelfs/share_data/ouyanglinke/OCR/OCR_validation_dataset.json"
|
||||
validation_dataset = "/mnt/petrelfs/share_data/ouyanglinke/OCR/OCR_validation_dataset_subset.json" # 测试
|
||||
save_path = "/mnt/petrelfs/share_data/ouyanglinke/OCR/OCR_val_result"
|
||||
validation(validation_dataset, pdf_bin_file_profile, pdf_model_dir, pdf_model_profile, save_path)
|
||||
@@ -0,0 +1,94 @@
|
||||
import os
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
import click
|
||||
import json
|
||||
from loguru import logger
|
||||
|
||||
from libs.commons import join_path, parse_aws_param, parse_bucket_key, read_file
|
||||
from mkcontent import mk_mm_markdown, mk_nlp_markdown
|
||||
from pdf_parse_by_model import parse_pdf_by_model
|
||||
|
||||
|
||||
|
||||
def main(s3_pdf_path: str, s3_pdf_profile: str, pdf_model_path: str, pdf_model_profile: str, start_page_num=0, debug_mode=True):
|
||||
""" """
|
||||
pth = Path(s3_pdf_path)
|
||||
book_name = pth.name
|
||||
# book_name = "".join(os.path.basename(s3_pdf_path).split(".")[0:-1])
|
||||
save_tmp_path = os.path.join(os.path.dirname(__file__), "..", "..","tmp", "unittest")
|
||||
save_path = join_path(save_tmp_path, "md")
|
||||
text_content_save_path = f"{save_path}/{book_name}/book.md"
|
||||
# metadata_save_path = f"{save_path}/{book_name}/metadata.json"
|
||||
|
||||
try:
|
||||
paras_dict = parse_pdf_by_model(
|
||||
s3_pdf_path, s3_pdf_profile, pdf_model_path, save_path, book_name, pdf_model_profile, start_page_num, debug_mode=debug_mode
|
||||
)
|
||||
parent_dir = os.path.dirname(text_content_save_path)
|
||||
if not os.path.exists(parent_dir):
|
||||
os.makedirs(parent_dir)
|
||||
|
||||
if not paras_dict.get('need_drop'):
|
||||
markdown_content = mk_mm_markdown(paras_dict)
|
||||
else:
|
||||
markdown_content = paras_dict['drop_reason']
|
||||
|
||||
with open(text_content_save_path, "w", encoding="utf-8") as f:
|
||||
f.write(markdown_content)
|
||||
|
||||
except Exception as e:
|
||||
print(f"ERROR: {s3_pdf_path}, {e}", file=sys.stderr)
|
||||
logger.exception(e)
|
||||
|
||||
|
||||
@click.command()
|
||||
@click.option("--pdf-file-path", help="s3上pdf文件的路径")
|
||||
@click.option("--save-path", help="解析出来的图片,文本的保存父目录")
|
||||
def main_shell(pdf_file_path: str, save_path: str):
|
||||
# pdf_bin_file_path = "s3://llm-raw-snew/llm-raw-scihub/scimag07865000-07865999/10.1007/"
|
||||
pdf_bin_file_parent_path = "s3://llm-raw-snew/llm-raw-scihub/"
|
||||
pdf_bin_file_profile = "s2"
|
||||
pdf_model_parent_dir = "s3://llm-pdf-text/eval_1k/layout_res/"
|
||||
pdf_model_profile = "langchao"
|
||||
|
||||
p = Path(pdf_file_path)
|
||||
pdf_parent_path = p.parent
|
||||
pdf_file_name = p.name # pdf文件名字,含后缀
|
||||
pdf_bin_file_path = join_path(pdf_bin_file_parent_path, pdf_parent_path)
|
||||
pdf_model_dir = join_path(pdf_model_parent_dir, pdf_parent_path)
|
||||
|
||||
main(
|
||||
join_path(pdf_bin_file_path, pdf_file_name),
|
||||
pdf_bin_file_profile,
|
||||
join_path(pdf_model_dir, pdf_file_name),
|
||||
pdf_model_profile,
|
||||
save_path,
|
||||
)
|
||||
|
||||
|
||||
@click.command()
|
||||
@click.option("--pdf-dir", help="s3上pdf文件的路径")
|
||||
@click.option("--model-dir", help="s3上pdf文件的路径")
|
||||
@click.option("--start-page-num", default=0, help="从第几页开始解析")
|
||||
def main_shell2(pdf_dir: str, model_dir: str,start_page_num: int):
|
||||
# 先扫描所有的pdf目录里的文件名字
|
||||
pdf_dir = Path(pdf_dir)
|
||||
model_dir = Path(model_dir)
|
||||
|
||||
if pdf_dir.is_file():
|
||||
pdf_file_names = [pdf_dir.name]
|
||||
pdf_dir = pdf_dir.parent
|
||||
else:
|
||||
pdf_file_names = [f.name for f in pdf_dir.glob("*.pdf")]
|
||||
|
||||
for pdf_file in pdf_file_names:
|
||||
pdf_file_path = os.path.join(pdf_dir, pdf_file)
|
||||
model_file_path = os.path.join(model_dir, pdf_file)
|
||||
main(pdf_file_path, None, model_file_path, None, start_page_num)
|
||||
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main_shell2()
|
||||
@@ -0,0 +1,107 @@
|
||||
import os
|
||||
import collections # 统计库
|
||||
import re # 正则
|
||||
from libs.commons import fitz # pyMuPDF库
|
||||
import json # json
|
||||
|
||||
|
||||
def calculate_overlapRatio_between_rect1_and_rect2(L1: float, U1: float, R1: float, D1: float, L2: float, U2: float, R2: float, D2: float) -> (float, float):
|
||||
# 计算两个rect,重叠面积各占2个rect面积的比例
|
||||
if min(R1, R2) < max(L1, L2) or min(D1, D2) < max(U1, U2):
|
||||
return 0, 0
|
||||
square_1 = (R1 - L1) * (D1 - U1)
|
||||
square_2 = (R2 - L2) * (D2 - U2)
|
||||
if square_1 == 0 or square_2 == 0:
|
||||
return 0, 0
|
||||
square_overlap = (min(R1, R2) - max(L1, L2)) * (min(D1, D2) - max(U1, U2))
|
||||
return square_overlap / square_1, square_overlap / square_2
|
||||
|
||||
|
||||
def evaluate_pdf_layout(page_ID: int, page: fitz.Page, json_from_DocXchain_obj: dict):
|
||||
"""
|
||||
:param page_ID: int类型,当前page在当前pdf文档中是第page_D页。
|
||||
:param page :fitz读取的当前页的内容
|
||||
:param res_dir_path: str类型,是每一个pdf文档,在当前.py文件的目录下生成一个与pdf文档同名的文件夹,res_dir_path就是文件夹的dir
|
||||
:param json_from_DocXchain_obj: dict类型,把pdf文档送入DocXChain模型中后,提取bbox,结果保存到pdf文档同名文件夹下的 page_ID.json文件中了。json_from_DocXchain_obj就是打开后的dict
|
||||
"""
|
||||
DPI = 72 # use this resolution
|
||||
pix = page.get_pixmap(dpi=DPI)
|
||||
pageL = 0
|
||||
pageR = int(pix.w)
|
||||
pageU = 0
|
||||
pageD = int(pix.h)
|
||||
|
||||
#--------- 通过json_from_DocXchain来获取 title ---------#
|
||||
title_bbox_from_DocXChain = []
|
||||
|
||||
xf_json = json_from_DocXchain_obj
|
||||
width_from_json = xf_json['page_info']['width']
|
||||
height_from_json = xf_json['page_info']['height']
|
||||
LR_scaleRatio = width_from_json / (pageR - pageL)
|
||||
UD_scaleRatio = height_from_json / (pageD - pageU)
|
||||
|
||||
# {0: 'title', # 标题
|
||||
# 1: 'figure', # 图片
|
||||
# 2: 'plain text', # 文本
|
||||
# 3: 'header', # 页眉
|
||||
# 4: 'page number', # 页码
|
||||
# 5: 'footnote', # 脚注
|
||||
# 6: 'footer', # 页脚
|
||||
# 7: 'table', # 表格
|
||||
# 8: 'table caption', # 表格描述
|
||||
# 9: 'figure caption', # 图片描述
|
||||
# 10: 'equation', # 公式
|
||||
# 11: 'full column', # 单栏
|
||||
# 12: 'sub column', # 多栏
|
||||
# 13: 'embedding', # 嵌入公式
|
||||
# 14: 'isolated'} # 单行公式
|
||||
LOSS_THRESHOLD = 2000 # 经验值
|
||||
fullColumn_bboxs = []
|
||||
subColumn_bboxs = []
|
||||
plainText_bboxs = []
|
||||
#### read information of plain text
|
||||
for xf in xf_json['layout_dets']:
|
||||
L = xf['poly'][0] / LR_scaleRatio
|
||||
U = xf['poly'][1] / UD_scaleRatio
|
||||
R = xf['poly'][2] / LR_scaleRatio
|
||||
D = xf['poly'][5] / UD_scaleRatio
|
||||
L, R = min(L, R), max(L, R)
|
||||
U, D = min(U, D), max(U, D)
|
||||
if xf['category_id'] == 2:
|
||||
plainText_bboxs.append((L, U, R, D))
|
||||
#### read information of column
|
||||
for xf in xf_json['subfield_dets']:
|
||||
L = xf['poly'][0] / LR_scaleRatio
|
||||
U = xf['poly'][1] / UD_scaleRatio
|
||||
R = xf['poly'][2] / LR_scaleRatio
|
||||
D = xf['poly'][5] / UD_scaleRatio
|
||||
L, R = min(L, R), max(L, R)
|
||||
U, D = min(U, D), max(U, D)
|
||||
if xf['category_id'] == 11:
|
||||
fullColumn_bboxs.append((L, U, R, D))
|
||||
elif xf['category_id'] == 12:
|
||||
subColumn_bboxs.append((L, U, R, D))
|
||||
|
||||
|
||||
curPage_loss = 0 # 当前页的loss
|
||||
fail_cnt = 0 # Text文本块没被圈到的情形。
|
||||
for L, U, R, D in plainText_bboxs:
|
||||
find = False
|
||||
for L2, U2, R2, D2 in (fullColumn_bboxs + subColumn_bboxs):
|
||||
ratio_1, _ = calculate_overlapRatio_between_rect1_and_rect2(L, U, R, D, L2, U2, R2, D2)
|
||||
if ratio_1 >= 0.9:
|
||||
loss_1 = (L + R) / 2 - (L2 + R2) / 2
|
||||
loss_2 = L - L2
|
||||
cur_loss = min(abs(loss_1), abs(loss_2))
|
||||
curPage_loss += cur_loss
|
||||
find = True
|
||||
break
|
||||
if find == False:
|
||||
fail_cnt += 1
|
||||
|
||||
isSimpleLayout_flag = False
|
||||
if fail_cnt == 0 and len(fullColumn_bboxs) <= 1 and len(subColumn_bboxs) <= 2:
|
||||
if curPage_loss <= LOSS_THRESHOLD:
|
||||
isSimpleLayout_flag = True
|
||||
|
||||
return isSimpleLayout_flag, len(fullColumn_bboxs), len(subColumn_bboxs), curPage_loss
|
||||
@@ -0,0 +1,345 @@
|
||||
from libs.commons import fitz
|
||||
from typing import List
|
||||
|
||||
|
||||
def show_image(item, title=""):
|
||||
"""Display a pixmap.
|
||||
|
||||
Just to display Pixmap image of "item" - ignore the man behind the curtain.
|
||||
|
||||
Args:
|
||||
item: any PyMuPDF object having a "get_pixmap" method.
|
||||
title: a string to be used as image title
|
||||
|
||||
Generates an RGB Pixmap from item using a constant DPI and using matplotlib
|
||||
to show it inline of the notebook.
|
||||
"""
|
||||
DPI = 150 # use this resolution
|
||||
import numpy as np
|
||||
import matplotlib.pyplot as plt
|
||||
|
||||
# %matplotlib inline
|
||||
pix = item.get_pixmap(dpi=DPI)
|
||||
img = np.ndarray([pix.h, pix.w, 3], dtype=np.uint8, buffer=pix.samples_mv)
|
||||
plt.figure(dpi=DPI) # set the figure's DPI
|
||||
plt.title(title) # set title of image
|
||||
_ = plt.imshow(img, extent=(0, pix.w * 72 / DPI, pix.h * 72 / DPI, 0))
|
||||
|
||||
|
||||
def calculate_overlapRatio_between_line1_and_line2(L1: float, R1: float, L2: float, R2: float) -> (float, float):
|
||||
# 计算两个line,重叠line各占2个line长度的比例
|
||||
if max(L1, L2) > min(R1, R2):
|
||||
return 0, 0
|
||||
if L1 == R1 or L2 == R2:
|
||||
return 0, 0
|
||||
overlap_line = min(R1, R2) - max(L1, L2)
|
||||
return overlap_line / (R1 - L1), overlap_line / (R2 - L2)
|
||||
|
||||
|
||||
def get_targetAxis_and_splitAxis(page_ID: int, page: fitz.Page, columnNumber: int, textBboxs: List[(float, float, float, float)]) -> (List[float], List[float]):
|
||||
"""
|
||||
param: page: fitz解析出来的格式
|
||||
param: columnNumber: Text的列数
|
||||
param: textBboxs: 文本块list。 [(L, U, R, D), ... ]
|
||||
return:
|
||||
|
||||
"""
|
||||
INF = 10 ** 9
|
||||
pageL, pageU, pageR, pageD = INF, INF, 0, 0
|
||||
for L, U, R, D in textBboxs:
|
||||
assert L <= R and U <= D
|
||||
pageL = min(pageL, L)
|
||||
pageR = max(pageR, R)
|
||||
pageU = min(pageU, U)
|
||||
pageD = max(pageD, D)
|
||||
|
||||
pageWidth = pageR - pageL
|
||||
pageHeight = pageD - pageU
|
||||
pageL -= pageWidth / 10 # 10是经验值
|
||||
pageR += pageWidth / 10
|
||||
pageU -= pageHeight / 10
|
||||
pageD += pageHeight / 10
|
||||
pageWidth = pageR - pageL
|
||||
pageHeight = pageD - pageU
|
||||
|
||||
x_targetAxis = []
|
||||
x_splitAxis = []
|
||||
for i in range(0, columnNumber * 2 + 1):
|
||||
if i & 1:
|
||||
x_targetAxis.append(pageL + pageWidth / (2 * columnNumber) * i)
|
||||
else:
|
||||
x_splitAxis.append(pageL + pageWidth / (2 * columnNumber) * i)
|
||||
|
||||
# # 可视化:分列的外框
|
||||
# path_bbox = []
|
||||
# N = len(x_targetAxis)
|
||||
# for i in range(N):
|
||||
# L, R = x_splitAxis[i], x_splitAxis[i + 1]
|
||||
# path_bbox.append((L, pageU, R, pageD))
|
||||
# shape = page.new_shape()
|
||||
# # iterate over the bboxes
|
||||
# color_map = [fitz.pdfcolor["red"], fitz.pdfcolor["blue"], fitz.pdfcolor["yellow"], fitz.pdfcolor["black"], fitz.pdfcolor["green"], fitz.pdfcolor["brown"]]
|
||||
# for i, rect in enumerate(path_bbox):
|
||||
# # if i < 20:
|
||||
# # continue
|
||||
# shape.draw_rect(rect) # draw a border
|
||||
# shape.insert_text(Point(rect[0], rect[1])+(5, 15), str(i), color=fitz.pdfcolor["blue"])
|
||||
# shape.finish(color=color_map[i%len(color_map)])
|
||||
# # shape.finish(color=fitz.pdfcolor["blue"])
|
||||
# shape.commit() # store to the page
|
||||
|
||||
# # if i == 3:
|
||||
# # print(rect)
|
||||
# # break
|
||||
# # print(rect)
|
||||
# show_image(page, f"Table & Header BBoxes")
|
||||
|
||||
return x_targetAxis, x_splitAxis
|
||||
|
||||
|
||||
def calculate_loss(page_ID: int, x_targetAxis: List[float], x_splitAxis: List[float], textBboxs: List[(float, float, float, float)]) -> (float, bool):
|
||||
INF = 10 ** 9
|
||||
|
||||
# page_artbox = page.artbox
|
||||
# pageL, pageU, pageR, pageD = page_artbox[0], page_artbox[1], page_artbox[2], page_artbox[3]
|
||||
|
||||
pageL, pageU, pageR, pageD = INF, INF, 0, 0
|
||||
for L, U, R, D in textBboxs:
|
||||
assert L <= R and U <= D
|
||||
pageL = min(pageL, L)
|
||||
pageR = max(pageR, R)
|
||||
pageU = min(pageU, U)
|
||||
pageD = max(pageD, D)
|
||||
|
||||
pageWidth = pageR - pageL
|
||||
pageHeight = pageD - pageU
|
||||
pageL -= pageWidth / 10
|
||||
pageR += pageWidth / 10
|
||||
pageU -= pageHeight / 10
|
||||
pageD += pageHeight / 10
|
||||
pageWidth = pageR - pageL
|
||||
pageHeight = pageD - pageU
|
||||
|
||||
col_N = len(x_targetAxis) # 列数
|
||||
col_texts_mid = [[] for _ in range(col_N)]
|
||||
col_texts_LR = [[] for _ in range(col_N)]
|
||||
|
||||
oneLocateLoss_mid = 0
|
||||
oneLocateLoss_LR = 0
|
||||
oneLocateCnt_mid = 0 # 完美在一列中的个数
|
||||
oneLocateCnt_LR = 0
|
||||
oneLocateSquare_mid = 0.0 # 完美在一列的面积
|
||||
oneLocateSquare_LR = 0.0
|
||||
|
||||
multiLocateLoss_mid = 0
|
||||
multiLocateLoss_LR = 0
|
||||
multiLocateCnt_mid = 0 # 在多列中的个数
|
||||
multiLocateCnt_LR = 0
|
||||
multiLocateSquare_mid = 0.0 # 在多列中的面积
|
||||
multiLocateSquare_LR = 0.0
|
||||
|
||||
allLocateLoss_mid = 0
|
||||
allLocateLoss_LR = 0
|
||||
allLocateCnt_mid = 0 # 横跨页面的大框的个数
|
||||
allLocateCnt_LR = 0
|
||||
allLocateSquare_mid = 0.0 # 横跨整个页面的个数
|
||||
allLocateSquare_LR = 0.0
|
||||
|
||||
isSimpleCondition = True # 就1个。2种方式,只要有一种情况不规整,就是不规整。
|
||||
colID_Textcnt_mid = [0 for _ in range(col_N)] # 每一列中有多少个Text块,根据mid判断的
|
||||
colID_Textcnt_LR = [0 for _ in range(col_N)] # 每一列中有多少个Text块,根据区间边界判断
|
||||
|
||||
allLocateBboxs_mid = [] # 跨整页的,bbox
|
||||
allLocateBboxs_LR = []
|
||||
non_allLocateBboxs_mid = []
|
||||
non_allLocateBboxs_LR = [] # 不在单独某一列,但又不是全列
|
||||
for L, U, R, D in textBboxs:
|
||||
if D - U < 40: # 现在还没拼接好。先简单这样过滤页眉。也会牺牲一些很窄的长条
|
||||
continue
|
||||
if R - L < 40:
|
||||
continue
|
||||
located_cols_mid = []
|
||||
located_cols_LR = []
|
||||
for col_ID in range(col_N):
|
||||
if col_N == 1:
|
||||
located_cols_mid.append(col_ID)
|
||||
located_cols_LR.append(col_ID)
|
||||
else:
|
||||
if L <= x_targetAxis[col_ID] <= R:
|
||||
located_cols_mid.append(col_ID)
|
||||
if calculate_overlapRatio_between_line1_and_line2(x_splitAxis[col_ID], x_splitAxis[col_ID + 1], L, R)[0] >= 0.2:
|
||||
located_cols_LR.append(col_ID)
|
||||
|
||||
if len(located_cols_mid) == col_N:
|
||||
allLocateBboxs_mid.append((L, U, R, D))
|
||||
else:
|
||||
non_allLocateBboxs_mid.append((L, U, R, D))
|
||||
if len(located_cols_LR) == col_N:
|
||||
allLocateBboxs_LR.append((L, U, R, D))
|
||||
else:
|
||||
non_allLocateBboxs_LR.append((L, U, R, D))
|
||||
|
||||
allLocateBboxs_mid.sort(key=lambda LURD: (LURD[1], LURD[0]))
|
||||
non_allLocateBboxs_mid.sort(key=lambda LURD: (LURD[1], LURD[0]))
|
||||
allLocateBboxs_LR.sort(key=lambda LURD: (LURD[1], LURD[0]))
|
||||
non_allLocateBboxs_LR.sort(key=lambda LURD: (LURD[1], LURD[0]))
|
||||
|
||||
# --------------------判断,是不是有标题类的小块,掺杂在一列的pdf页面里。-------------#
|
||||
isOneClumn = False
|
||||
under_cnt = 0
|
||||
under_square = 0.0
|
||||
before_cnt = 0
|
||||
before_square = 0.0
|
||||
for nL, nU, nR, nD in non_allLocateBboxs_mid:
|
||||
cnt = 0
|
||||
for L, U, R, D in allLocateBboxs_mid:
|
||||
if nD <= U:
|
||||
cnt += 1
|
||||
if cnt >= 1:
|
||||
before_cnt += cnt
|
||||
before_square += (R - L) * (D - U) * cnt
|
||||
else:
|
||||
under_cnt += 1
|
||||
under_square += (R - L) * (D - U) * cnt
|
||||
|
||||
if (before_square + under_square) != 0 and before_square / (before_square + under_square) >= 0.2:
|
||||
isOneClumn = True
|
||||
|
||||
if isOneClumn == True and col_N != 1:
|
||||
return INF, False
|
||||
if isOneClumn == True and col_N == 1:
|
||||
return 0, True
|
||||
#### 根据边界的统计情况,再判断一次
|
||||
isOneClumn = False
|
||||
under_cnt = 0
|
||||
under_square = 0.0
|
||||
before_cnt = 0
|
||||
before_square = 0.0
|
||||
for nL, nU, nR, nD in non_allLocateBboxs_LR:
|
||||
cnt = 0
|
||||
for L, U, R, D in allLocateBboxs_LR:
|
||||
if nD <= U:
|
||||
cnt += 1
|
||||
if cnt >= 1:
|
||||
before_cnt += cnt
|
||||
before_square += (R - L) * (D - U) * cnt
|
||||
else:
|
||||
under_cnt += 1
|
||||
under_square += (R - L) * (D - U) * cnt
|
||||
|
||||
if (before_square + under_square) != 0 and before_square / (before_square + under_square) >= 0.2:
|
||||
isOneClumn = True
|
||||
|
||||
if isOneClumn == True and col_N != 1:
|
||||
return INF, False
|
||||
if isOneClumn == True and col_N == 1:
|
||||
return 0, True
|
||||
|
||||
for L, U, R, D in textBboxs:
|
||||
assert L < R and U < D, 'There is an error on bbox of text when calculate loss!'
|
||||
|
||||
# 简单排除页眉、迷你小块
|
||||
# if (D - U) < pageHeight / 15 < 40 or (R - L) < pageWidth / 8:
|
||||
if (D - U) < 40:
|
||||
continue
|
||||
if (R - L) < 40:
|
||||
continue
|
||||
mid = (L + R) / 2
|
||||
located_cols_mid = [] # 在哪一列里,根据中点来判断
|
||||
located_cols_LR = [] # 在哪一列里,根据边界判断
|
||||
for col_ID in range(col_N):
|
||||
if col_N == 1:
|
||||
located_cols_mid.append(col_ID)
|
||||
else:
|
||||
# 根据中点判断
|
||||
if L <= x_targetAxis[col_ID] <= R:
|
||||
located_cols_mid.append(col_ID)
|
||||
# 根据边界判断
|
||||
if calculate_overlapRatio_between_line1_and_line2(x_splitAxis[col_ID], x_splitAxis[col_ID + 1], L, R)[0] >= 0.2:
|
||||
located_cols_LR.append(col_ID)
|
||||
|
||||
## 1列的情形
|
||||
if col_N == 1:
|
||||
oneLocateLoss_mid += abs(mid - x_targetAxis[located_cols_mid[0]]) * (D - U) * (R - L)
|
||||
# oneLocateLoss_mid += abs(L - x_splitAxis[located_cols[0]]) * (D - U) * (R - L)
|
||||
oneLocateLoss_LR += abs(L - x_splitAxis[located_cols_mid[0]]) * (D - U) * (R - L)
|
||||
oneLocateCnt_mid += 1
|
||||
oneLocateSquare_mid += (D - U) * (R - L)
|
||||
## 多列的情形
|
||||
else:
|
||||
######## 根据mid判断
|
||||
if len(located_cols_mid) == 1:
|
||||
oneLocateLoss_mid += abs(mid - x_targetAxis[located_cols_mid[0]]) * (D - U) * (R - L)
|
||||
# oneLocateLoss_mid += abs(L - x_splitAxis[located_cols[0]]) * (D - U) * (R - L)
|
||||
oneLocateCnt_mid += 1
|
||||
oneLocateSquare_mid += (D - U) * (R - L)
|
||||
elif 1 <= len(located_cols_mid) < col_N:
|
||||
ll, rr = located_cols_mid[0], located_cols_mid[-1]
|
||||
# multiLocateLoss_mid += abs(mid - (x_targetAxis[ll] + x_targetAxis[rr]) / 2) * (D - U) * (R - L)
|
||||
multiLocateLoss_mid += abs(mid - x_targetAxis[ll]) * (D - U) * (R - L)
|
||||
# multiLocateLoss_mid += abs(mid - (pageL + pageR) / 2) * (D - U) * (R - L)
|
||||
multiLocateCnt_mid += 1
|
||||
multiLocateSquare_mid += (D - U) * (R - L)
|
||||
isSimpleCondition = False
|
||||
else:
|
||||
allLocateLoss_mid += abs(mid - (pageR + pageL) / 2) * (D - U) * (R - L)
|
||||
allLocateCnt_mid += 1
|
||||
allLocateSquare_mid += (D - U) * (R - L)
|
||||
isSimpleCondition = False
|
||||
|
||||
######## 根据区间的边界判断
|
||||
if len(located_cols_LR) == 1:
|
||||
oneLocateLoss_LR += abs(mid - x_targetAxis[located_cols_LR[0]]) * (D - U) * (R - L)
|
||||
# oneLocateLoss_LR += abs(L - x_splitAxis[located_cols_LR[0]]) * (D - U) * (R - L)
|
||||
oneLocateCnt_LR += 1
|
||||
oneLocateSquare_LR += (D - U) * (R - L)
|
||||
elif 1 <= len(located_cols_LR) < col_N:
|
||||
ll, rr = located_cols_LR[0], located_cols_LR[-1]
|
||||
# multiLocateLoss_LR += abs(mid - (x_targetAxis[ll] + x_targetAxis[rr]) / 2) * (D - U) * (R - L)
|
||||
multiLocateLoss_LR += abs(mid - x_targetAxis[ll]) * (D - U) * (R - L)
|
||||
# multiLocateLoss_LR += abs(mid - (pageL + pageR) / 2) * (D - U) * (R - L)
|
||||
multiLocateCnt_LR += 1
|
||||
multiLocateSquare_LR += (D - U) * (R - L)
|
||||
isSimpleCondition = False
|
||||
else:
|
||||
allLocateLoss_LR += abs(mid - (pageR + pageL) / 2) * (D - U) * (R - L)
|
||||
allLocateCnt_LR += 1
|
||||
allLocateSquare_LR += (D - U) * (R - L)
|
||||
isSimpleCondition = False
|
||||
|
||||
tot_TextCnt = oneLocateCnt_mid + multiLocateCnt_mid + allLocateCnt_mid
|
||||
tot_TextSquare = oneLocateSquare_mid + multiLocateSquare_mid + allLocateSquare_mid
|
||||
|
||||
# 1列的情形
|
||||
if tot_TextSquare != 0 and allLocateSquare_mid / tot_TextSquare >= 0.85 and col_N == 1:
|
||||
return 0, True
|
||||
|
||||
# 多列的情形
|
||||
|
||||
# if col_N >= 2:
|
||||
# if allLocateCnt >= 1:
|
||||
# oneLocateLoss_mid += ((pageR - pageL)) * oneLocateCnt_mid
|
||||
# multiLocateLoss_mid += ((pageR - pageL) ) * multiLocateCnt_mid
|
||||
# else:
|
||||
# if multiLocateCnt_mid >= 1:
|
||||
# oneLocateLoss_mid += ((pageR - pageL)) * oneLocateCnt_mid
|
||||
totLoss_mid = oneLocateLoss_mid + multiLocateLoss_mid + allLocateLoss_mid
|
||||
totLoss_LR = oneLocateCnt_LR + multiLocateCnt_LR + allLocateLoss_LR
|
||||
return totLoss_mid + totLoss_LR, isSimpleCondition
|
||||
|
||||
|
||||
def get_columnNumber(page_ID: int, page: fitz.Page, textBboxs) -> (int, float):
|
||||
columnNumber_loss = dict()
|
||||
columnNumber_isSimpleCondition = dict()
|
||||
#### 枚举列数
|
||||
for columnNumber in range(1, 5):
|
||||
# print('---------{}--------'.format(columnNumber))
|
||||
x_targetAxis, x_splitAxis = get_targetAxis_and_splitAxis(page_ID, page, columnNumber, textBboxs)
|
||||
loss, isSimpleCondition = calculate_loss(page_ID, x_targetAxis, x_splitAxis, textBboxs)
|
||||
columnNumber_loss[columnNumber] = loss
|
||||
columnNumber_isSimpleCondition[columnNumber] = isSimpleCondition
|
||||
|
||||
col_idxs = [i for i in range(1, len(columnNumber_loss) + 1)]
|
||||
col_idxs.sort(key=lambda i: (columnNumber_loss[i], i))
|
||||
|
||||
return col_idxs, columnNumber_loss, columnNumber_isSimpleCondition
|
||||
@@ -0,0 +1,104 @@
|
||||
import os
|
||||
import collections # 统计库
|
||||
import re # 正则
|
||||
from libs.commons import fitz # pyMuPDF库
|
||||
import json # json
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
|
||||
def parse_equations(page_ID: int, page: fitz.Page, json_from_DocXchain_obj: dict):
|
||||
"""
|
||||
:param page_ID: int类型,当前page在当前pdf文档中是第page_D页。
|
||||
:param page :fitz读取的当前页的内容
|
||||
:param res_dir_path: str类型,是每一个pdf文档,在当前.py文件的目录下生成一个与pdf文档同名的文件夹,res_dir_path就是文件夹的dir
|
||||
:param json_from_DocXchain_obj: dict类型,把pdf文档送入DocXChain模型中后,提取bbox,结果保存到pdf文档同名文件夹下的 page_ID.json文件中了。json_from_DocXchain_obj就是打开后的dict
|
||||
"""
|
||||
DPI = 72 # use this resolution
|
||||
pix = page.get_pixmap(dpi=DPI)
|
||||
pageL = 0
|
||||
pageR = int(pix.w)
|
||||
pageU = 0
|
||||
pageD = int(pix.h)
|
||||
|
||||
|
||||
#--------- 通过json_from_DocXchain来获取 table ---------#
|
||||
equationEmbedding_from_DocXChain_bboxs = []
|
||||
equationIsolated_from_DocXChain_bboxs = []
|
||||
|
||||
xf_json = json_from_DocXchain_obj
|
||||
width_from_json = xf_json['page_info']['width']
|
||||
height_from_json = xf_json['page_info']['height']
|
||||
LR_scaleRatio = width_from_json / (pageR - pageL)
|
||||
UD_scaleRatio = height_from_json / (pageD - pageU)
|
||||
|
||||
for xf in xf_json['layout_dets']:
|
||||
# {0: 'title', 1: 'figure', 2: 'plain text', 3: 'header', 4: 'page number', 5: 'footnote', 6: 'footer', 7: 'table', 8: 'table caption', 9: 'figure caption', 10: 'equation', 11: 'full column', 12: 'sub column'}
|
||||
L = xf['poly'][0] / LR_scaleRatio
|
||||
U = xf['poly'][1] / UD_scaleRatio
|
||||
R = xf['poly'][2] / LR_scaleRatio
|
||||
D = xf['poly'][5] / UD_scaleRatio
|
||||
# L += pageL # 有的页面,artBox偏移了。不在(0,0)
|
||||
# R += pageL
|
||||
# U += pageU
|
||||
# D += pageU
|
||||
L, R = min(L, R), max(L, R)
|
||||
U, D = min(U, D), max(U, D)
|
||||
# equation
|
||||
img_suffix = f"{page_ID}_{int(L)}_{int(U)}_{int(R)}_{int(D)}"
|
||||
if xf['category_id'] == 13 and xf['score'] >= 0.3:
|
||||
latex_text = xf.get("latex", "EmptyInlineEquationResult")
|
||||
debugable_latex_text = f"{latex_text}|{img_suffix}"
|
||||
equationEmbedding_from_DocXChain_bboxs.append((L, U, R, D, latex_text))
|
||||
if xf['category_id'] == 14 and xf['score'] >= 0.3:
|
||||
latex_text = xf.get("latex", "EmptyInterlineEquationResult")
|
||||
debugable_latex_text = f"{latex_text}|{img_suffix}"
|
||||
equationIsolated_from_DocXChain_bboxs.append((L, U, R, D, latex_text))
|
||||
|
||||
#---------------------------------------- 排序,编号,保存 -----------------------------------------#
|
||||
equationIsolated_from_DocXChain_bboxs.sort(key = lambda LURD: (LURD[1], LURD[0]))
|
||||
equationIsolated_from_DocXChain_bboxs.sort(key = lambda LURD: (LURD[1], LURD[0]))
|
||||
|
||||
equationEmbedding_from_DocXChain_names = []
|
||||
equationEmbedding_ID = 0
|
||||
|
||||
equationIsolated_from_DocXChain_names = []
|
||||
equationIsolated_ID = 0
|
||||
|
||||
for L, U, R, D, _ in equationEmbedding_from_DocXChain_bboxs:
|
||||
if not(L < R and U < D):
|
||||
continue
|
||||
try:
|
||||
# cur_equation = page.get_pixmap(clip=(L,U,R,D))
|
||||
new_equation_name = "equationEmbedding_{}_{}.png".format(page_ID, equationEmbedding_ID) # 公式name
|
||||
# cur_equation.save(res_dir_path + '/' + new_equation_name) # 把公式存出在新建的文件夹,并命名
|
||||
equationEmbedding_from_DocXChain_names.append(new_equation_name) # 把公式的名字存在list中,方便在md中插入引用
|
||||
equationEmbedding_ID += 1
|
||||
except:
|
||||
pass
|
||||
|
||||
for L, U, R, D, _ in equationIsolated_from_DocXChain_bboxs:
|
||||
if not(L < R and U < D):
|
||||
continue
|
||||
try:
|
||||
# cur_equation = page.get_pixmap(clip=(L,U,R,D))
|
||||
new_equation_name = "equationEmbedding_{}_{}.png".format(page_ID, equationIsolated_ID) # 公式name
|
||||
# cur_equation.save(res_dir_path + '/' + new_equation_name) # 把公式存出在新建的文件夹,并命名
|
||||
equationIsolated_from_DocXChain_names.append(new_equation_name) # 把公式的名字存在list中,方便在md中插入引用
|
||||
equationIsolated_ID += 1
|
||||
except:
|
||||
pass
|
||||
|
||||
equationEmbedding_from_DocXChain_bboxs.sort(key = lambda LURD: (LURD[1], LURD[0]))
|
||||
equationIsolated_from_DocXChain_bboxs.sort(key = lambda LURD: (LURD[1], LURD[0]))
|
||||
|
||||
|
||||
"""根据pdf可视区域,调整bbox的坐标"""
|
||||
cropbox = page.cropbox
|
||||
if cropbox[0]!=page.rect[0] or cropbox[1]!=page.rect[1]:
|
||||
for eq_box in equationEmbedding_from_DocXChain_bboxs:
|
||||
eq_box = [eq_box[0]+cropbox[0], eq_box[1]+cropbox[1], eq_box[2]+cropbox[0], eq_box[3]+cropbox[1], eq_box[4]]
|
||||
for eq_box in equationIsolated_from_DocXChain_bboxs:
|
||||
eq_box = [eq_box[0]+cropbox[0], eq_box[1]+cropbox[1], eq_box[2]+cropbox[0], eq_box[3]+cropbox[1], eq_box[4]]
|
||||
|
||||
return equationEmbedding_from_DocXChain_bboxs, equationIsolated_from_DocXChain_bboxs
|
||||
@@ -0,0 +1,650 @@
|
||||
import os
|
||||
import collections # 统计库
|
||||
import re
|
||||
from libs.boxbase import _is_in_or_part_overlap # 正则
|
||||
from libs.commons import fitz # pyMuPDF库
|
||||
import json # json
|
||||
|
||||
|
||||
#--------------------------------------- Tool Functions --------------------------------------#
|
||||
# 正则化,输入文本,输出只保留a-z,A-Z,0-9
|
||||
def remove_special_chars(s: str) -> str:
|
||||
pattern = r"[^a-zA-Z0-9]"
|
||||
res = re.sub(pattern, "", s)
|
||||
return res
|
||||
|
||||
def check_rect1_sameWith_rect2(L1: float, U1: float, R1: float, D1: float, L2: float, U2: float, R2: float, D2: float) -> bool:
|
||||
# 判断rect1和rect2是否一模一样
|
||||
return L1 == L2 and U1 == U2 and R1 == R2 and D1 == D2
|
||||
|
||||
def check_rect1_contains_rect2(L1: float, U1: float, R1: float, D1: float, L2: float, U2: float, R2: float, D2: float) -> bool:
|
||||
# 判断rect1包含了rect2
|
||||
return (L1 <= L2 <= R2 <= R1) and (U1 <= U2 <= D2 <= D1)
|
||||
|
||||
def check_rect1_overlaps_rect2(L1: float, U1: float, R1: float, D1: float, L2: float, U2: float, R2: float, D2: float) -> bool:
|
||||
# 判断rect1与rect2是否存在重叠(只有一条边重叠,也算重叠)
|
||||
return max(L1, L2) <= min(R1, R2) and max(U1, U2) <= min(D1, D2)
|
||||
|
||||
def calculate_overlapRatio_between_rect1_and_rect2(L1: float, U1: float, R1: float, D1: float, L2: float, U2: float, R2: float, D2: float) -> (float, float):
|
||||
# 计算两个rect,重叠面积各占2个rect面积的比例
|
||||
if min(R1, R2) < max(L1, L2) or min(D1, D2) < max(U1, U2):
|
||||
return 0, 0
|
||||
square_1 = (R1 - L1) * (D1 - U1)
|
||||
square_2 = (R2 - L2) * (D2 - U2)
|
||||
if square_1 == 0 or square_2 == 0:
|
||||
return 0, 0
|
||||
square_overlap = (min(R1, R2) - max(L1, L2)) * (min(D1, D2) - max(U1, U2))
|
||||
return square_overlap / square_1, square_overlap / square_2
|
||||
|
||||
def calculate_overlapRatio_between_line1_and_line2(L1: float, R1: float, L2: float, R2: float) -> (float, float):
|
||||
# 计算两个line,重叠区间各占2个line长度的比例
|
||||
if max(L1, L2) > min(R1, R2):
|
||||
return 0, 0
|
||||
if L1 == R1 or L2 == R2:
|
||||
return 0, 0
|
||||
overlap_line = min(R1, R2) - max(L1, L2)
|
||||
return overlap_line / (R1 - L1), overlap_line / (R2 - L2)
|
||||
|
||||
|
||||
# 判断rect其实是一条line
|
||||
def check_rect_isLine(L: float, U: float, R: float, D: float) -> bool:
|
||||
width = R - L
|
||||
height = D - U
|
||||
if width <= 3 or height <= 3:
|
||||
return True
|
||||
if width / height >= 30 or height / width >= 30:
|
||||
return True
|
||||
|
||||
|
||||
|
||||
def parse_images(page_ID: int, page: fitz.Page, json_from_DocXchain_obj: dict, junk_img_bojids=[]):
|
||||
"""
|
||||
:param page_ID: int类型,当前page在当前pdf文档中是第page_D页。
|
||||
:param page :fitz读取的当前页的内容
|
||||
:param res_dir_path: str类型,是每一个pdf文档,在当前.py文件的目录下生成一个与pdf文档同名的文件夹,res_dir_path就是文件夹的dir
|
||||
:param json_from_DocXchain_obj: dict类型,把pdf文档送入DocXChain模型中后,提取bbox,结果保存到pdf文档同名文件夹下的 page_ID.json文件中了。json_from_DocXchain_obj就是打开后的dict
|
||||
"""
|
||||
#### 通过fitz获取page信息
|
||||
## 超越边界
|
||||
DPI = 72 # use this resolution
|
||||
pix = page.get_pixmap(dpi=DPI)
|
||||
pageL = 0
|
||||
pageR = int(pix.w)
|
||||
pageU = 0
|
||||
pageD = int(pix.h)
|
||||
|
||||
#----------------- 保存每一个文本块的LURD ------------------#
|
||||
textLine_blocks = []
|
||||
blocks = page.get_text(
|
||||
"dict",
|
||||
flags=fitz.TEXTFLAGS_TEXT,
|
||||
#clip=clip,
|
||||
)["blocks"]
|
||||
for i in range(len(blocks)):
|
||||
bbox = blocks[i]['bbox']
|
||||
# print(bbox)
|
||||
for tt in blocks[i]['lines']:
|
||||
# 当前line
|
||||
cur_line_bbox = None # 当前line,最右侧的section的bbox
|
||||
for xf in tt['spans']:
|
||||
L, U, R, D = xf['bbox']
|
||||
L, R = min(L, R), max(L, R)
|
||||
U, D = min(U, D), max(U, D)
|
||||
textLine_blocks.append((L, U, R, D))
|
||||
textLine_blocks.sort(key = lambda LURD: (LURD[1], LURD[0]))
|
||||
|
||||
|
||||
#---------------------------------------------- 保存img --------------------------------------------------#
|
||||
raw_imgs = page.get_images() # 获取所有的图片
|
||||
imgs = []
|
||||
img_names = [] # 保存图片的名字,方便在md中插入引用
|
||||
img_bboxs = [] # 保存图片的location信息。
|
||||
img_visited = [] # 记忆化,记录该图片是否在md中已经插入过了
|
||||
img_ID = 0
|
||||
|
||||
## 获取、保存每张img的location信息(x1, y1, x2, y2, UL, DR坐标)
|
||||
for i in range(len(raw_imgs)):
|
||||
# 如果图片在junklist中则跳过
|
||||
if raw_imgs[i][0] in junk_img_bojids:
|
||||
continue
|
||||
else:
|
||||
try:
|
||||
tt = page.get_image_rects(raw_imgs[i][0], transform = True)
|
||||
|
||||
rec = tt[0][0]
|
||||
L, U, R, D = int(rec[0]), int(rec[1]), int(rec[2]), int(rec[3])
|
||||
|
||||
L, R = min(L, R), max(L, R)
|
||||
U, D = min(U, D), max(U, D)
|
||||
if not(pageL <= L < R <= pageR and pageU <= U < D <= pageD):
|
||||
continue
|
||||
if pageL == L and R == pageR:
|
||||
continue
|
||||
if pageU == U and D == pageD:
|
||||
continue
|
||||
# pix1 = page.get_Pixmap(clip=(L,U,R,D))
|
||||
new_img_name = "{}_{}.png".format(page_ID, i) # 图片name
|
||||
# pix1.save(res_dir_path + '/' + new_img_name) # 把图片存出在新建的文件夹,并命名
|
||||
img_names.append(new_img_name)
|
||||
img_bboxs.append((L, U, R, D))
|
||||
img_visited.append(False)
|
||||
imgs.append(raw_imgs[i])
|
||||
except:
|
||||
continue
|
||||
|
||||
#-------- 如果img之间有重叠。说明获取的img大小有问题,位置也不一定对。就扔掉--------#
|
||||
imgs_ok = [True for _ in range(len(imgs))]
|
||||
for i in range(len(imgs)):
|
||||
L1, U1, R1, D1 = img_bboxs[i]
|
||||
for j in range(i + 1, len(imgs)):
|
||||
L2, U2, R2, D2 = img_bboxs[j]
|
||||
ratio_1, ratio_2 = calculate_overlapRatio_between_rect1_and_rect2(L1, U1, R1, D1, L2, U2, R2, D2)
|
||||
s1 = abs(R1 - L1) * abs(D1 - U1)
|
||||
s2 = abs(R2 - L2) * abs(D2 - U2)
|
||||
if ratio_1 > 0 and ratio_2 > 0:
|
||||
if ratio_1 == 1 and ratio_2 > 0.8:
|
||||
imgs_ok[i] = False
|
||||
elif ratio_1 > 0.8 and ratio_2 == 1:
|
||||
imgs_ok[j] = False
|
||||
elif s1 > 20000 and s2 > 20000 and ratio_1 > 0.4 and ratio_2 > 0.4:
|
||||
imgs_ok[i] = False
|
||||
imgs_ok[j] = False
|
||||
elif s1 / s2 > 5 and ratio_2 > 0.5:
|
||||
imgs_ok[j] = False
|
||||
elif s2 / s1 > 5 and ratio_1 > 0.5:
|
||||
imgs_ok[i] = False
|
||||
|
||||
imgs = [imgs[i] for i in range(len(imgs)) if imgs_ok[i] == True]
|
||||
img_names = [img_names[i] for i in range(len(imgs)) if imgs_ok[i] == True]
|
||||
img_bboxs = [img_bboxs[i] for i in range(len(imgs)) if imgs_ok[i] == True]
|
||||
img_visited = [img_visited[i] for i in range(len(imgs)) if imgs_ok[i] == True]
|
||||
#*******************************************************************************#
|
||||
|
||||
#---------------------------------------- 通过fitz提取svg的信息 -----------------------------------------#
|
||||
#
|
||||
svgs = page.get_drawings()
|
||||
#------------ preprocess, check一些大框,看是否是合理的 ----------#
|
||||
## 去重。有时候会遇到rect1和rect2是完全一样的情形。
|
||||
svg_rect_visited = set()
|
||||
available_svgIdx = []
|
||||
for i in range(len(svgs)):
|
||||
L, U, R, D = svgs[i]['rect'].irect
|
||||
L, R = min(L, R), max(L, R)
|
||||
U, D = min(U, D), max(U, D)
|
||||
tt = (L, U, R, D)
|
||||
if tt not in svg_rect_visited:
|
||||
svg_rect_visited.add(tt)
|
||||
available_svgIdx.append(i)
|
||||
|
||||
svgs = [svgs[i] for i in available_svgIdx] # 去重后,有效的svgs
|
||||
svg_childs = [[] for _ in range(len(svgs))]
|
||||
svg_parents = [[] for _ in range(len(svgs))]
|
||||
svg_overlaps = [[] for _ in range(len(svgs))] #svg_overlaps[i]是一个list,存的是与svg_i有重叠的svg的index。e.g., svg_overlaps[0] = [1, 2, 7, 9]
|
||||
svg_visited = [False for _ in range(len(svgs))]
|
||||
svg_exceedPage = [0 for _ in range(len(svgs))] # 是否超越边界(artbox),很大,但一般是一个svg的底。
|
||||
|
||||
|
||||
for i in range(len(svgs)):
|
||||
L, U, R, D = svgs[i]['rect'].irect
|
||||
ratio_1, ratio_2 = calculate_overlapRatio_between_rect1_and_rect2(L, U, R, D, pageL, pageU, pageR, pageD)
|
||||
if (pageL + 20 < L <= R < pageR - 20) and (pageU + 20 < U <= D < pageD - 20):
|
||||
if ratio_2 >= 0.7:
|
||||
svg_exceedPage[i] += 4
|
||||
else:
|
||||
if L <= pageL:
|
||||
svg_exceedPage[i] += 1
|
||||
if pageR <= R:
|
||||
svg_exceedPage[i] += 1
|
||||
if U <= pageU:
|
||||
svg_exceedPage[i] += 1
|
||||
if pageD <= D:
|
||||
svg_exceedPage[i] += 1
|
||||
|
||||
#### 如果有≥2个的超边界的框,就不要手写规则判断svg了。很难写对。
|
||||
if len([x for x in svg_exceedPage if x >= 1]) >= 2:
|
||||
svgs = []
|
||||
svg_childs = []
|
||||
svg_parents = []
|
||||
svg_overlaps = []
|
||||
svg_visited = []
|
||||
svg_exceedPage = []
|
||||
|
||||
#---------------------------- build graph ----------------------------#
|
||||
for i, p in enumerate(svgs):
|
||||
L1, U1, R1, D1 = svgs[i]["rect"].irect
|
||||
for j in range(len(svgs)):
|
||||
if i == j:
|
||||
continue
|
||||
L2, U2, R2, D2 = svgs[j]["rect"].irect
|
||||
## 包含
|
||||
if check_rect1_contains_rect2(L1, U1, R1, D1, L2, U2, R2, D2) == True:
|
||||
svg_childs[i].append(j)
|
||||
svg_parents[j].append(i)
|
||||
else:
|
||||
## 交叉
|
||||
if check_rect1_overlaps_rect2(L1, U1, R1, D1, L2, U2, R2, D2) == True:
|
||||
svg_overlaps[i].append(j)
|
||||
|
||||
#---------------- 确定最终的svg。连通块儿的外围 -------------------#
|
||||
eps_ERROR = 5 # 给识别出的svg,四周留白(为了防止pyMuPDF的rect不准)
|
||||
svg_ID = 0
|
||||
svg_final_names = []
|
||||
svg_final_bboxs = []
|
||||
svg_final_visited = [] # 为下面,text识别左准备。作用同img_visited
|
||||
|
||||
svg_idxs = [i for i in range(len(svgs))]
|
||||
svg_idxs.sort(key = lambda i: -(svgs[i]['rect'].irect[2] - svgs[i]['rect'].irect[0]) * (svgs[i]['rect'].irect[3] - svgs[i]['rect'].irect[1])) # 按照面积,从大到小排序
|
||||
|
||||
for i in svg_idxs:
|
||||
if svg_visited[i] == True:
|
||||
continue
|
||||
svg_visited[i] = True
|
||||
L, U, R, D = svgs[i]['rect'].irect
|
||||
width = R - L
|
||||
height = D - U
|
||||
if check_rect_isLine(L, U, R, D) == True:
|
||||
svg_visited[i] = False
|
||||
continue
|
||||
# if i == 4:
|
||||
# print(i, L, U, R, D)
|
||||
# print(svg_parents[i])
|
||||
|
||||
cur_block_element_cnt = 0 # 当前要判定为svg的区域中,有多少elements,最外围的最大svg框除外。
|
||||
if len(svg_parents[i]) == 0:
|
||||
## 是个普通框的情形
|
||||
cur_block_element_cnt += len(svg_childs[i])
|
||||
if svg_exceedPage[i] == 0:
|
||||
## 误差。可能已经包含在某个框里面了
|
||||
neglect_flag = False
|
||||
for pL, pU, pR, pD in svg_final_bboxs:
|
||||
if pL <= L <= R <= pR and pU <= U <= D <= pD:
|
||||
neglect_flag = True
|
||||
break
|
||||
if neglect_flag == True:
|
||||
continue
|
||||
|
||||
## 搜索连通域, bfs+记忆化
|
||||
q = collections.deque()
|
||||
for j in svg_overlaps[i]:
|
||||
q.append(j)
|
||||
while q:
|
||||
j = q.popleft()
|
||||
svg_visited[j] = True
|
||||
L2, U2, R2, D2 = svgs[j]['rect'].irect
|
||||
# width2 = R2 - L2
|
||||
# height2 = D2 - U2
|
||||
# if width2 <= 2 or height2 <= 2 or (height2 / width2) >= 30 or (width2 / height2) >= 30:
|
||||
# continue
|
||||
L = min(L, L2)
|
||||
R = max(R, R2)
|
||||
U = min(U, U2)
|
||||
D = max(D, D2)
|
||||
cur_block_element_cnt += 1
|
||||
cur_block_element_cnt += len(svg_childs[j])
|
||||
for k in svg_overlaps[j]:
|
||||
if svg_visited[k] == False and svg_exceedPage[k] == 0:
|
||||
svg_visited[k] = True
|
||||
q.append(k)
|
||||
elif svg_exceedPage[i] <= 2:
|
||||
## 误差。可能已经包含在某个svg_final_bbox框里面了
|
||||
neglect_flag = False
|
||||
for sL, sU, sR, sD in svg_final_bboxs:
|
||||
if sL <= L <= R <= sR and sU <= U <= D <= sD:
|
||||
neglect_flag = True
|
||||
break
|
||||
if neglect_flag == True:
|
||||
continue
|
||||
|
||||
L, U, R, D = pageR, pageD, pageL, pageU
|
||||
## 所有孩子元素的最大边界
|
||||
for j in svg_childs[i]:
|
||||
if svg_visited[j] == True:
|
||||
continue
|
||||
if svg_exceedPage[j] >= 1:
|
||||
continue
|
||||
svg_visited[j] = True #### 这个位置考虑一下
|
||||
L2, U2, R2, D2 = svgs[j]['rect'].irect
|
||||
L = min(L, L2)
|
||||
R = max(R, R2)
|
||||
U = min(U, U2)
|
||||
D = max(D, D2)
|
||||
cur_block_element_cnt += 1
|
||||
|
||||
# 如果是条line,就不用保存了
|
||||
if check_rect_isLine(L, U, R, D) == True:
|
||||
continue
|
||||
# 如果当前的svg,连2个elements都没有,就不用保存了
|
||||
if cur_block_element_cnt < 3:
|
||||
continue
|
||||
|
||||
## 当前svg,框住了多少文本框。如果框多了,可能就是错了
|
||||
contain_textLineBlock_cnt = 0
|
||||
for L2, U2, R2, D2 in textLine_blocks:
|
||||
if check_rect1_contains_rect2(L, U, R, D, L2, U2, R2, D2) == True:
|
||||
contain_textLineBlock_cnt += 1
|
||||
if contain_textLineBlock_cnt >= 10:
|
||||
continue
|
||||
|
||||
# L -= eps_ERROR * 2
|
||||
# U -= eps_ERROR
|
||||
# R += eps_ERROR * 2
|
||||
# D += eps_ERROR
|
||||
# # cur_svg = page.get_pixmap(matrix=fitz.Identity, dpi=None, colorspace=fitz.csRGB, clip=(U,L,R,D), alpha=False, annots=True)
|
||||
# cur_svg = page.get_pixmap(clip=(L,U,R,D))
|
||||
new_svg_name = "svg_{}_{}.png".format(page_ID, svg_ID) # 图片name
|
||||
# cur_svg.save(res_dir_path + '/' + new_svg_name) # 把图片存出在新建的文件夹,并命名
|
||||
svg_final_names.append(new_svg_name) # 把图片的名字存在list中,方便在md中插入引用
|
||||
svg_final_bboxs.append((L, U, R, D))
|
||||
svg_final_visited.append(False)
|
||||
svg_ID += 1
|
||||
|
||||
## 识别出的svg,可能有 包含,相邻的情形。需要进一步合并
|
||||
svg_idxs = [i for i in range(len(svg_final_bboxs))]
|
||||
svg_idxs.sort(key = lambda i: (svg_final_bboxs[i][1], svg_final_bboxs[i][0])) # (U, L)
|
||||
svg_final_names_2 = []
|
||||
svg_final_bboxs_2 = []
|
||||
svg_final_visited_2 = [] # 为下面,text识别左准备。作用同img_visited
|
||||
svg_ID_2 = 0
|
||||
for i in range(len(svg_final_bboxs)):
|
||||
L1, U1, R1, D1 = svg_final_bboxs[i]
|
||||
for j in range(i + 1, len(svg_final_bboxs)):
|
||||
L2, U2, R2, D2 = svg_final_bboxs[j]
|
||||
# 如果 rect1包含了rect2
|
||||
if check_rect1_contains_rect2(L1, U1, R1, D1, L2, U2, R2, D2) == True:
|
||||
svg_final_visited[j] = True
|
||||
continue
|
||||
# 水平并列
|
||||
ratio_1, ratio_2 = calculate_overlapRatio_between_line1_and_line2(U1, D1, U2, D2)
|
||||
if ratio_1 >= 0.7 and ratio_2 >= 0.7:
|
||||
if abs(L2 - R1) >= 20:
|
||||
continue
|
||||
LL = min(L1, L2)
|
||||
UU = min(U1, U2)
|
||||
RR = max(R1, R2)
|
||||
DD = max(D1, D2)
|
||||
svg_final_bboxs[i] = (LL, UU, RR, DD)
|
||||
svg_final_visited[j] = True
|
||||
continue
|
||||
# 竖直并列
|
||||
ratio_1, ratio_2 = calculate_overlapRatio_between_line1_and_line2(L1, R2, L2, R2)
|
||||
if ratio_1 >= 0.7 and ratio_2 >= 0.7:
|
||||
if abs(U2 - D1) >= 20:
|
||||
continue
|
||||
LL = min(L1, L2)
|
||||
UU = min(U1, U2)
|
||||
RR = max(R1, R2)
|
||||
DD = max(D1, D2)
|
||||
svg_final_bboxs[i] = (LL, UU, RR, DD)
|
||||
svg_final_visited[j] = True
|
||||
|
||||
for i in range(len(svg_final_bboxs)):
|
||||
if svg_final_visited[i] == False:
|
||||
L, U, R, D = svg_final_bboxs[i]
|
||||
svg_final_bboxs_2.append((L, U, R, D))
|
||||
|
||||
L -= eps_ERROR * 2
|
||||
U -= eps_ERROR
|
||||
R += eps_ERROR * 2
|
||||
D += eps_ERROR
|
||||
# cur_svg = page.get_pixmap(clip=(L,U,R,D))
|
||||
new_svg_name = "svg_{}_{}.png".format(page_ID, svg_ID_2) # 图片name
|
||||
# cur_svg.save(res_dir_path + '/' + new_svg_name) # 把图片存出在新建的文件夹,并命名
|
||||
svg_final_names_2.append(new_svg_name) # 把图片的名字存在list中,方便在md中插入引用
|
||||
svg_final_bboxs_2.append((L, U, R, D))
|
||||
svg_final_visited_2.append(False)
|
||||
svg_ID_2 += 1
|
||||
|
||||
## svg收尾。识别为drawing,但是在上面没有拼成一张图的。
|
||||
# 有收尾才comprehensive
|
||||
# xxxx
|
||||
# xxxx
|
||||
# xxxx
|
||||
# xxxx
|
||||
|
||||
|
||||
#--------- 通过json_from_DocXchain来获取,figure, table, equation的bbox ---------#
|
||||
figure_bbox_from_DocXChain = []
|
||||
|
||||
figure_from_DocXChain_visited = [] # 记忆化
|
||||
figure_bbox_from_DocXChain_overlappedRatio = []
|
||||
|
||||
figure_only_from_DocXChain_bboxs = [] # 存储
|
||||
figure_only_from_DocXChain_names = []
|
||||
figure_only_from_DocXChain_visited = []
|
||||
figure_only_ID = 0
|
||||
|
||||
xf_json = json_from_DocXchain_obj
|
||||
width_from_json = xf_json['page_info']['width']
|
||||
height_from_json = xf_json['page_info']['height']
|
||||
LR_scaleRatio = width_from_json / (pageR - pageL)
|
||||
UD_scaleRatio = height_from_json / (pageD - pageU)
|
||||
|
||||
for xf in xf_json['layout_dets']:
|
||||
# {0: 'title', 1: 'figure', 2: 'plain text', 3: 'header', 4: 'page number', 5: 'footnote', 6: 'footer', 7: 'table', 8: 'table caption', 9: 'figure caption', 10: 'equation', 11: 'full column', 12: 'sub column'}
|
||||
L = xf['poly'][0] / LR_scaleRatio
|
||||
U = xf['poly'][1] / UD_scaleRatio
|
||||
R = xf['poly'][2] / LR_scaleRatio
|
||||
D = xf['poly'][5] / UD_scaleRatio
|
||||
# L += pageL # 有的页面,artBox偏移了。不在(0,0)
|
||||
# R += pageL
|
||||
# U += pageU
|
||||
# D += pageU
|
||||
L, R = min(L, R), max(L, R)
|
||||
U, D = min(U, D), max(U, D)
|
||||
# figure
|
||||
if xf["category_id"] == 1 and xf['score'] >= 0.3:
|
||||
figure_bbox_from_DocXChain.append((L, U, R, D))
|
||||
figure_from_DocXChain_visited.append(False)
|
||||
figure_bbox_from_DocXChain_overlappedRatio.append(0.0)
|
||||
|
||||
#---------------------- 比对上面识别出来的img,svg 与DocXChain给的figure -----------------------#
|
||||
|
||||
## 比对imgs
|
||||
for i, b1 in enumerate(figure_bbox_from_DocXChain):
|
||||
# print('--------- DocXChain的图片', b1)
|
||||
L1, U1, R1, D1 = b1
|
||||
for b2 in img_bboxs:
|
||||
# print('-------- igms得到的图', b2)
|
||||
L2, U2, R2, D2 = b2
|
||||
s1 = abs(R1 - L1) * abs(D1 - U1)
|
||||
s2 = abs(R2 - L2) * abs(D2 - U2)
|
||||
# 相同
|
||||
if check_rect1_sameWith_rect2(L1, U1, R1, D1, L2, U2, R2, D2) == True:
|
||||
figure_from_DocXChain_visited[i] = True
|
||||
# 包含
|
||||
elif check_rect1_contains_rect2(L1, U1, R1, D1, L2, U2, R2, D2) == True:
|
||||
if s2 / s1 > 0.8:
|
||||
figure_from_DocXChain_visited[i] = True
|
||||
elif check_rect1_contains_rect2(L2, U2, R2, D2, L1, U1, R1, D1) == True:
|
||||
if s1 / s2 > 0.8:
|
||||
figure_from_DocXChain_visited[i] = True
|
||||
else:
|
||||
# 重叠了相当一部分
|
||||
# print('进入第3部分')
|
||||
ratio_1, ratio_2 = calculate_overlapRatio_between_rect1_and_rect2(L1, U1, R1, D1, L2, U2, R2, D2)
|
||||
if (ratio_1 >= 0.6 and ratio_2 >= 0.6) or (ratio_1 >= 0.8 and s1/s2>0.8) or (ratio_2 >= 0.8 and s2/s1>0.8):
|
||||
figure_from_DocXChain_visited[i] = True
|
||||
else:
|
||||
figure_bbox_from_DocXChain_overlappedRatio[i] += ratio_1
|
||||
# print('图片的重叠率是{}'.format(ratio_1))
|
||||
|
||||
|
||||
## 比对svgs
|
||||
svg_final_bboxs_2_badIdxs = []
|
||||
for i, b1 in enumerate(figure_bbox_from_DocXChain):
|
||||
L1, U1, R1, D1 = b1
|
||||
for j, b2 in enumerate(svg_final_bboxs_2):
|
||||
L2, U2, R2, D2 = b2
|
||||
s1 = abs(R1 - L1) * abs(D1 - U1)
|
||||
s2 = abs(R2 - L2) * abs(D2 - U2)
|
||||
# 相同
|
||||
if check_rect1_sameWith_rect2(L1, U1, R1, D1, L2, U2, R2, D2) == True:
|
||||
figure_from_DocXChain_visited[i] = True
|
||||
# 包含
|
||||
elif check_rect1_contains_rect2(L1, U1, R1, D1, L2, U2, R2, D2) == True:
|
||||
figure_from_DocXChain_visited[i] = True
|
||||
elif check_rect1_contains_rect2(L2, U2, R2, D2, L1, U1, R1, D1) == True:
|
||||
if s1 / s2 > 0.7:
|
||||
figure_from_DocXChain_visited[i] = True
|
||||
else:
|
||||
svg_final_bboxs_2_badIdxs.append(j) # svg丢弃。用DocXChain的结果。
|
||||
else:
|
||||
# 重叠了相当一部分
|
||||
ratio_1, ratio_2 = calculate_overlapRatio_between_rect1_and_rect2(L1, U1, R1, D1, L2, U2, R2, D2)
|
||||
if (ratio_1 >= 0.5 and ratio_2 >= 0.5) or (min(ratio_1, ratio_2) >= 0.4 and max(ratio_1, ratio_2) >= 0.6):
|
||||
figure_from_DocXChain_visited[i] = True
|
||||
else:
|
||||
figure_bbox_from_DocXChain_overlappedRatio[i] += ratio_1
|
||||
|
||||
# 丢掉错误的svg
|
||||
svg_final_bboxs_2 = [svg_final_bboxs_2[i] for i in range(len(svg_final_bboxs_2)) if i not in set(svg_final_bboxs_2_badIdxs)]
|
||||
|
||||
for i in range(len(figure_from_DocXChain_visited)):
|
||||
if figure_bbox_from_DocXChain_overlappedRatio[i] >= 0.7:
|
||||
figure_from_DocXChain_visited[i] = True
|
||||
|
||||
# DocXChain识别出来的figure,但是没被保存的。
|
||||
for i in range(len(figure_from_DocXChain_visited)):
|
||||
if figure_from_DocXChain_visited[i] == False:
|
||||
figure_from_DocXChain_visited[i] = True
|
||||
cur_bbox = figure_bbox_from_DocXChain[i]
|
||||
# cur_figure = page.get_pixmap(clip=cur_bbox)
|
||||
new_figure_name = "figure_only_{}_{}.png".format(page_ID, figure_only_ID) # 图片name
|
||||
# cur_figure.save(res_dir_path + '/' + new_figure_name) # 把图片存出在新建的文件夹,并命名
|
||||
figure_only_from_DocXChain_names.append(new_figure_name) # 把图片的名字存在list中,方便在md中插入引用
|
||||
figure_only_from_DocXChain_bboxs.append(cur_bbox)
|
||||
figure_only_from_DocXChain_visited.append(False)
|
||||
figure_only_ID += 1
|
||||
|
||||
img_bboxs.sort(key = lambda LURD: (LURD[1], LURD[0]))
|
||||
svg_final_bboxs_2.sort(key = lambda LURD: (LURD[1], LURD[0]))
|
||||
figure_only_from_DocXChain_bboxs.sort(key = lambda LURD: (LURD[1], LURD[0]))
|
||||
curPage_all_fig_bboxs = img_bboxs + svg_final_bboxs + figure_only_from_DocXChain_bboxs
|
||||
|
||||
#--------------------------- 最后统一去重 -----------------------------------#
|
||||
curPage_all_fig_bboxs.sort(key = lambda LURD: ( (LURD[2]-LURD[0])*(LURD[3]-LURD[1]) , LURD[0], LURD[1]) )
|
||||
|
||||
#### 先考虑包含关系的小块
|
||||
final_duplicate = set()
|
||||
for i in range(len(curPage_all_fig_bboxs)):
|
||||
L1, U1, R1, D1 = curPage_all_fig_bboxs[i]
|
||||
for j in range(len(curPage_all_fig_bboxs)):
|
||||
if i == j:
|
||||
continue
|
||||
L2, U2, R2, D2 = curPage_all_fig_bboxs[j]
|
||||
s1 = abs(R1 - L1) * abs(D1 - U1)
|
||||
s2 = abs(R2 - L2) * abs(D2 - U2)
|
||||
if check_rect1_contains_rect2(L2, U2, R2, D2, L1, U1, R1, D1) == True:
|
||||
final_duplicate.add((L1, U1, R1, D1))
|
||||
else:
|
||||
ratio_1, ratio_2 = calculate_overlapRatio_between_rect1_and_rect2(L1, U1, R1, D1, L2, U2, R2, D2)
|
||||
if ratio_1 >= 0.8 and ratio_2 <= 0.6:
|
||||
final_duplicate.add((L1, U1, R1, D1))
|
||||
|
||||
curPage_all_fig_bboxs = [LURD for LURD in curPage_all_fig_bboxs if LURD not in final_duplicate]
|
||||
|
||||
#### 再考虑重叠关系的块
|
||||
final_duplicate = set()
|
||||
final_synthetic_bboxs = []
|
||||
for i in range(len(curPage_all_fig_bboxs)):
|
||||
L1, U1, R1, D1 = curPage_all_fig_bboxs[i]
|
||||
for j in range(len(curPage_all_fig_bboxs)):
|
||||
if i == j:
|
||||
continue
|
||||
L2, U2, R2, D2 = curPage_all_fig_bboxs[j]
|
||||
s1 = abs(R1 - L1) * abs(D1 - U1)
|
||||
s2 = abs(R2 - L2) * abs(D2 - U2)
|
||||
ratio_1, ratio_2 = calculate_overlapRatio_between_rect1_and_rect2(L1, U1, R1, D1, L2, U2, R2, D2)
|
||||
union_ok = False
|
||||
if (ratio_1 >= 0.8 and ratio_2 <= 0.6) or (ratio_1 > 0.6 and ratio_2 > 0.6):
|
||||
union_ok = True
|
||||
if (ratio_1 > 0.2 and s2 / s1 > 5):
|
||||
union_ok = True
|
||||
if (L1 <= (L2+R2)/2 <= R1) and (U1 <= (U2+D2)/2 <= D1):
|
||||
union_ok = True
|
||||
if (L2 <= (L1+R1)/2 <= R2) and (U2 <= (U1+D1)/2 <= D2):
|
||||
union_ok = True
|
||||
if union_ok == True:
|
||||
final_duplicate.add((L1, U1, R1, D1))
|
||||
final_duplicate.add((L2, U2, R2, D2))
|
||||
L3, U3, R3, D3 = min(L1, L2), min(U1, U2), max(R1, R2), max(D1, D2)
|
||||
final_synthetic_bboxs.append((L3, U3, R3, D3))
|
||||
|
||||
# print('---------- curPage_all_fig_bboxs ---------')
|
||||
# print(curPage_all_fig_bboxs)
|
||||
curPage_all_fig_bboxs = [b for b in curPage_all_fig_bboxs if b not in final_duplicate]
|
||||
final_synthetic_bboxs = list(set(final_synthetic_bboxs))
|
||||
|
||||
|
||||
## 再再考虑重叠关系。极端情况下会迭代式地2进1
|
||||
new_images = []
|
||||
droped_img_idx = []
|
||||
image_bboxes = [[b[0], b[1], b[2], b[3]] for b in final_synthetic_bboxs]
|
||||
for i in range(0, len(image_bboxes)):
|
||||
for j in range(i+1, len(image_bboxes)):
|
||||
if j not in droped_img_idx:
|
||||
L2, U2, R2, D2 = image_bboxes[j]
|
||||
s1 = abs(R1 - L1) * abs(D1 - U1)
|
||||
s2 = abs(R2 - L2) * abs(D2 - U2)
|
||||
ratio_1, ratio_2 = calculate_overlapRatio_between_rect1_and_rect2(L1, U1, R1, D1, L2, U2, R2, D2)
|
||||
union_ok = False
|
||||
if (ratio_1 >= 0.8 and ratio_2 <= 0.6) or (ratio_1 > 0.6 and ratio_2 > 0.6):
|
||||
union_ok = True
|
||||
if (ratio_1 > 0.2 and s2 / s1 > 5):
|
||||
union_ok = True
|
||||
if (L1 <= (L2+R2)/2 <= R1) and (U1 <= (U2+D2)/2 <= D1):
|
||||
union_ok = True
|
||||
if (L2 <= (L1+R1)/2 <= R2) and (U2 <= (U1+D1)/2 <= D2):
|
||||
union_ok = True
|
||||
if union_ok == True:
|
||||
# 合并
|
||||
image_bboxes[i][0], image_bboxes[i][1],image_bboxes[i][2],image_bboxes[i][3] = min(image_bboxes[i][0], image_bboxes[j][0]), min(image_bboxes[i][1], image_bboxes[j][1]), max(image_bboxes[i][2], image_bboxes[j][2]), max(image_bboxes[i][3], image_bboxes[j][3])
|
||||
droped_img_idx.append(j)
|
||||
|
||||
for i in range(0, len(image_bboxes)):
|
||||
if i not in droped_img_idx:
|
||||
new_images.append(image_bboxes[i])
|
||||
|
||||
|
||||
# find_union_FLAG = True
|
||||
# while find_union_FLAG == True:
|
||||
# find_union_FLAG = False
|
||||
# final_duplicate = set()
|
||||
# tmp = []
|
||||
# for i in range(len(final_synthetic_bboxs)):
|
||||
# L1, U1, R1, D1 = final_synthetic_bboxs[i]
|
||||
# for j in range(len(final_synthetic_bboxs)):
|
||||
# if i == j:
|
||||
# continue
|
||||
# L2, U2, R2, D2 = final_synthetic_bboxs[j]
|
||||
# s1 = abs(R1 - L1) * abs(D1 - U1)
|
||||
# s2 = abs(R2 - L2) * abs(D2 - U2)
|
||||
# ratio_1, ratio_2 = calculate_overlapRatio_between_rect1_and_rect2(L1, U1, R1, D1, L2, U2, R2, D2)
|
||||
# union_ok = False
|
||||
# if (ratio_1 >= 0.8 and ratio_2 <= 0.6) or (ratio_1 > 0.6 and ratio_2 > 0.6):
|
||||
# union_ok = True
|
||||
# if (ratio_1 > 0.2 and s2 / s1 > 5):
|
||||
# union_ok = True
|
||||
# if (L1 <= (L2+R2)/2 <= R1) and (U1 <= (U2+D2)/2 <= D1):
|
||||
# union_ok = True
|
||||
# if (L2 <= (L1+R1)/2 <= R2) and (U2 <= (U1+D1)/2 <= D2):
|
||||
# union_ok = True
|
||||
# if union_ok == True:
|
||||
# find_union_FLAG = True
|
||||
# final_duplicate.add((L1, U1, R1, D1))
|
||||
# final_duplicate.add((L2, U2, R2, D2))
|
||||
# L3, U3, R3, D3 = min(L1, L2), min(U1, U2), max(R1, R2), max(D1, D2)
|
||||
# tmp.append((L3, U3, R3, D3))
|
||||
# if find_union_FLAG == True:
|
||||
# tmp = list(set(tmp))
|
||||
# final_synthetic_bboxs = tmp[:]
|
||||
|
||||
|
||||
# curPage_all_fig_bboxs += final_synthetic_bboxs
|
||||
# print('--------- final synthetic')
|
||||
# print(final_synthetic_bboxs)
|
||||
#**************************************************************************#
|
||||
images1 = [[img[0], img[1], img[2], img[3]] for img in curPage_all_fig_bboxs]
|
||||
images = images1 + new_images
|
||||
return images
|
||||
|
||||
@@ -0,0 +1,78 @@
|
||||
import os
|
||||
import collections # 统计库
|
||||
import re # 正则
|
||||
from libs.commons import fitz # pyMuPDF库
|
||||
import json # json
|
||||
|
||||
|
||||
def parse_footers(page_ID: int, page: fitz.Page, json_from_DocXchain_obj: dict):
|
||||
"""
|
||||
:param page_ID: int类型,当前page在当前pdf文档中是第page_D页。
|
||||
:param page :fitz读取的当前页的内容
|
||||
:param res_dir_path: str类型,是每一个pdf文档,在当前.py文件的目录下生成一个与pdf文档同名的文件夹,res_dir_path就是文件夹的dir
|
||||
:param json_from_DocXchain_obj: dict类型,把pdf文档送入DocXChain模型中后,提取bbox,结果保存到pdf文档同名文件夹下的 page_ID.json文件中了。json_from_DocXchain_obj就是打开后的dict
|
||||
"""
|
||||
DPI = 72 # use this resolution
|
||||
pix = page.get_pixmap(dpi=DPI)
|
||||
pageL = 0
|
||||
pageR = int(pix.w)
|
||||
pageU = 0
|
||||
pageD = int(pix.h)
|
||||
|
||||
|
||||
#--------- 通过json_from_DocXchain来获取 footer ---------#
|
||||
footer_bbox_from_DocXChain = []
|
||||
|
||||
|
||||
xf_json = json_from_DocXchain_obj
|
||||
width_from_json = xf_json['page_info']['width']
|
||||
height_from_json = xf_json['page_info']['height']
|
||||
LR_scaleRatio = width_from_json / (pageR - pageL)
|
||||
UD_scaleRatio = height_from_json / (pageD - pageU)
|
||||
|
||||
# {0: 'title', # 标题
|
||||
# 1: 'figure', # 图片
|
||||
# 2: 'plain text', # 文本
|
||||
# 3: 'header', # 页眉
|
||||
# 4: 'page number', # 页码
|
||||
# 5: 'footnote', # 脚注
|
||||
# 6: 'footer', # 页脚
|
||||
# 7: 'table', # 表格
|
||||
# 8: 'table caption', # 表格描述
|
||||
# 9: 'figure caption', # 图片描述
|
||||
# 10: 'equation', # 公式
|
||||
# 11: 'full column', # 单栏
|
||||
# 12: 'sub column', # 多栏
|
||||
# 13: 'embedding', # 嵌入公式
|
||||
# 14: 'isolated'} # 单行公式
|
||||
for xf in xf_json['layout_dets']:
|
||||
L = xf['poly'][0] / LR_scaleRatio
|
||||
U = xf['poly'][1] / UD_scaleRatio
|
||||
R = xf['poly'][2] / LR_scaleRatio
|
||||
D = xf['poly'][5] / UD_scaleRatio
|
||||
# L += pageL # 有的页面,artBox偏移了。不在(0,0)
|
||||
# R += pageL
|
||||
# U += pageU
|
||||
# D += pageU
|
||||
L, R = min(L, R), max(L, R)
|
||||
U, D = min(U, D), max(U, D)
|
||||
if xf['category_id'] == 6 and xf['score'] >= 0.3:
|
||||
footer_bbox_from_DocXChain.append((L, U, R, D))
|
||||
|
||||
|
||||
footer_final_names = []
|
||||
footer_final_bboxs = []
|
||||
footer_ID = 0
|
||||
for L, U, R, D in footer_bbox_from_DocXChain:
|
||||
# cur_footer = page.get_pixmap(clip=(L,U,R,D))
|
||||
new_footer_name = "footer_{}_{}.png".format(page_ID, footer_ID) # 脚注name
|
||||
# cur_footer.save(res_dir_path + '/' + new_footer_name) # 把页脚存储在新建的文件夹,并命名
|
||||
footer_final_names.append(new_footer_name) # 把脚注的名字存在list中
|
||||
footer_final_bboxs.append((L, U, R, D))
|
||||
footer_ID += 1
|
||||
|
||||
|
||||
footer_final_bboxs.sort(key = lambda LURD: (LURD[1], LURD[0]))
|
||||
curPage_all_footer_bboxs = footer_final_bboxs
|
||||
return curPage_all_footer_bboxs
|
||||
|
||||
@@ -0,0 +1,173 @@
|
||||
import os
|
||||
from collections import Counter
|
||||
import re # 正则
|
||||
from libs.commons import fitz # pyMuPDF库
|
||||
import json # json
|
||||
|
||||
|
||||
def parse_footnotes_by_model(page_ID: int, page: fitz.Page, json_from_DocXchain_obj: dict, md_bookname_save_path, debug_mode=False):
|
||||
"""
|
||||
:param page_ID: int类型,当前page在当前pdf文档中是第page_D页。
|
||||
:param page :fitz读取的当前页的内容
|
||||
:param res_dir_path: str类型,是每一个pdf文档,在当前.py文件的目录下生成一个与pdf文档同名的文件夹,res_dir_path就是文件夹的dir
|
||||
:param json_from_DocXchain_obj: dict类型,把pdf文档送入DocXChain模型中后,提取bbox,结果保存到pdf文档同名文件夹下的 page_ID.json文件中了。json_from_DocXchain_obj就是打开后的dict
|
||||
"""
|
||||
DPI = 72 # use this resolution
|
||||
pix = page.get_pixmap(dpi=DPI)
|
||||
pageL = 0
|
||||
pageR = int(pix.w)
|
||||
pageU = 0
|
||||
pageD = int(pix.h)
|
||||
|
||||
|
||||
#--------- 通过json_from_DocXchain来获取 footnote ---------#
|
||||
footnote_bbox_from_DocXChain = []
|
||||
|
||||
xf_json = json_from_DocXchain_obj
|
||||
width_from_json = xf_json['page_info']['width']
|
||||
height_from_json = xf_json['page_info']['height']
|
||||
LR_scaleRatio = width_from_json / (pageR - pageL)
|
||||
UD_scaleRatio = height_from_json / (pageD - pageU)
|
||||
|
||||
# {0: 'title', # 标题
|
||||
# 1: 'figure', # 图片
|
||||
# 2: 'plain text', # 文本
|
||||
# 3: 'header', # 页眉
|
||||
# 4: 'page number', # 页码
|
||||
# 5: 'footnote', # 脚注
|
||||
# 6: 'footer', # 页脚
|
||||
# 7: 'table', # 表格
|
||||
# 8: 'table caption', # 表格描述
|
||||
# 9: 'figure caption', # 图片描述
|
||||
# 10: 'equation', # 公式
|
||||
# 11: 'full column', # 单栏
|
||||
# 12: 'sub column', # 多栏
|
||||
# 13: 'embedding', # 嵌入公式
|
||||
# 14: 'isolated'} # 单行公式
|
||||
for xf in xf_json['layout_dets']:
|
||||
L = xf['poly'][0] / LR_scaleRatio
|
||||
U = xf['poly'][1] / UD_scaleRatio
|
||||
R = xf['poly'][2] / LR_scaleRatio
|
||||
D = xf['poly'][5] / UD_scaleRatio
|
||||
# L += pageL # 有的页面,artBox偏移了。不在(0,0)
|
||||
# R += pageL
|
||||
# U += pageU
|
||||
# D += pageU
|
||||
L, R = min(L, R), max(L, R)
|
||||
U, D = min(U, D), max(U, D)
|
||||
# if xf['category_id'] == 5 and xf['score'] >= 0.3:
|
||||
if xf['category_id'] == 5 and xf['score'] >= 0.43: # 新的footnote阈值
|
||||
footnote_bbox_from_DocXChain.append((L, U, R, D))
|
||||
|
||||
|
||||
footnote_final_names = []
|
||||
footnote_final_bboxs = []
|
||||
footnote_ID = 0
|
||||
for L, U, R, D in footnote_bbox_from_DocXChain:
|
||||
if debug_mode:
|
||||
# cur_footnote = page.get_pixmap(clip=(L,U,R,D))
|
||||
new_footnote_name = "footnote_{}_{}.png".format(page_ID, footnote_ID) # 脚注name
|
||||
# cur_footnote.save(md_bookname_save_path + '/' + new_footnote_name) # 把脚注存储在新建的文件夹,并命名
|
||||
footnote_final_names.append(new_footnote_name) # 把脚注的名字存在list中
|
||||
footnote_final_bboxs.append((L, U, R, D))
|
||||
footnote_ID += 1
|
||||
|
||||
|
||||
footnote_final_bboxs.sort(key = lambda LURD: (LURD[1], LURD[0]))
|
||||
curPage_all_footnote_bboxs = footnote_final_bboxs
|
||||
return curPage_all_footnote_bboxs
|
||||
|
||||
|
||||
def need_remove(block):
|
||||
if 'lines' in block and len(block['lines']) > 0:
|
||||
# block中只有一行,且该行文本全是大写字母,或字体为粗体bold关键词,SB关键词,把这个block捞回来
|
||||
if len(block['lines']) == 1:
|
||||
if 'spans' in block['lines'][0] and len(block['lines'][0]['spans']) == 1:
|
||||
font_keywords = ['SB', 'bold', 'Bold']
|
||||
if block['lines'][0]['spans'][0]['text'].isupper() or any(keyword in block['lines'][0]['spans'][0]['font'] for keyword in font_keywords):
|
||||
return True
|
||||
for line in block['lines']:
|
||||
if 'spans' in line and len(line['spans']) > 0:
|
||||
for span in line['spans']:
|
||||
# 检测"keyword"是否在span中,忽略大小写
|
||||
if "keyword" in span['text'].lower():
|
||||
return True
|
||||
return False
|
||||
|
||||
def parse_footnotes_by_rule(remain_text_blocks, page_height, page_id, main_text_font):
|
||||
"""
|
||||
根据给定的文本块、页高和页码,解析出符合规则的脚注文本块,并返回其边界框。
|
||||
|
||||
Args:
|
||||
remain_text_blocks (list): 包含所有待处理的文本块的列表。
|
||||
page_height (float): 页面的高度。
|
||||
page_id (int): 页面的ID。
|
||||
|
||||
Returns:
|
||||
list: 符合规则的脚注文本块的边界框列表。
|
||||
|
||||
"""
|
||||
if page_id > 20:
|
||||
return []
|
||||
else:
|
||||
# 存储每一行的文本块大小的列表
|
||||
line_sizes = []
|
||||
# 存储每个文本块的平均行大小
|
||||
block_sizes = []
|
||||
# 存储每一行的字体信息
|
||||
# font_names = []
|
||||
font_names = Counter()
|
||||
if len(remain_text_blocks) > 0:
|
||||
for block in remain_text_blocks:
|
||||
block_line_sizes = []
|
||||
# block_fonts = []
|
||||
block_fonts = Counter()
|
||||
for line in block['lines']:
|
||||
# 提取每个span的size属性,并计算行大小
|
||||
span_sizes = [span['size'] for span in line['spans'] if 'size' in span]
|
||||
if span_sizes:
|
||||
line_size = sum(span_sizes) / len(span_sizes)
|
||||
line_sizes.append(line_size)
|
||||
block_line_sizes.append(line_size)
|
||||
span_font = [(span['font'], len(span['text'])) for span in line['spans'] if 'font' in span and len(span['text']) > 0]
|
||||
if span_font:
|
||||
# # todo main_text_font应该用基于字数最多的字体而不是span级别的统计
|
||||
# font_names.append(font_name for font_name in span_font)
|
||||
# block_fonts.append(font_name for font_name in span_font)
|
||||
for font, count in span_font:
|
||||
# font_names.extend([font] * count)
|
||||
# block_fonts.extend([font] * count)
|
||||
font_names[font] += count
|
||||
block_fonts[font] += count
|
||||
if block_line_sizes:
|
||||
# 计算文本块的平均行大小
|
||||
block_size = sum(block_line_sizes) / len(block_line_sizes)
|
||||
# block_font = collections.Counter(block_fonts).most_common(1)[0][0]
|
||||
block_font = block_fonts.most_common(1)[0][0]
|
||||
block_sizes.append((block, block_size, block_font))
|
||||
|
||||
# 计算main_text_size
|
||||
main_text_size = Counter(line_sizes).most_common(1)[0][0]
|
||||
# 计算main_text_font
|
||||
# main_text_font = collections.Counter(font_names).most_common(1)[0][0]
|
||||
# main_text_font = font_names.most_common(1)[0][0]
|
||||
# 删除一些可能被误识别为脚注的文本块
|
||||
block_sizes = [(block, block_size, block_font) for block, block_size, block_font in block_sizes if not need_remove(block)]
|
||||
|
||||
# 检测footnote_block 并返回 footnote_bboxes
|
||||
# footnote_bboxes = [block['bbox'] for block, block_size, block_font in block_sizes if
|
||||
# block['bbox'][1] > page_height * 0.6 and block_size < main_text_size
|
||||
# and (len(block['lines']) < 5 or block_font != main_text_font)]
|
||||
# and len(block['lines']) < 5]
|
||||
footnote_bboxes = [block['bbox'] for block, block_size, block_font in block_sizes if
|
||||
block['bbox'][1] > page_height * 0.6 and
|
||||
sum([block_size < main_text_size,
|
||||
len(block['lines']) < 5,
|
||||
block_font != main_text_font]) >= 2]
|
||||
|
||||
return footnote_bboxes
|
||||
else:
|
||||
return []
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,673 @@
|
||||
import io
|
||||
import re
|
||||
import os
|
||||
import json
|
||||
from libs.boxbase import _is_in_or_part_overlap, calculate_overlap_area_2_minbox_area_ratio
|
||||
from libs.commons import fitz
|
||||
from fitz import Point
|
||||
from pprint import pprint
|
||||
import pickle
|
||||
import collections
|
||||
from typing import List
|
||||
|
||||
|
||||
def calculate_overlapRatio_between_rect1_and_rect2(L1: float, U1: float, R1: float, D1: float, L2: float, U2: float, R2: float, D2: float) -> (float, float):
|
||||
# 计算两个rect,重叠面积各占2个rect面积的比例
|
||||
if min(R1, R2) < max(L1, L2) or min(D1, D2) < max(U1, U2):
|
||||
return 0, 0
|
||||
square_1 = (R1 - L1) * (D1 - U1)
|
||||
square_2 = (R2 - L2) * (D2 - U2)
|
||||
if square_1 == 0 or square_2 == 0:
|
||||
return 0, 0
|
||||
square_overlap = (min(R1, R2) - max(L1, L2)) * (min(D1, D2) - max(U1, U2))
|
||||
return square_overlap / square_1, square_overlap / square_2
|
||||
|
||||
def calculate_overlapRatio_between_line1_and_line2(L1: float, R1: float, L2: float, R2: float) -> (float, float):
|
||||
# 计算两个line,重叠区间各占2个line长度的比例
|
||||
if max(L1, L2) > min(R1, R2):
|
||||
return 0, 0
|
||||
if L1 == R1 or L2 == R2:
|
||||
return 0, 0
|
||||
overlap_line = min(R1, R2) - max(L1, L2)
|
||||
return overlap_line / (R1 - L1), overlap_line / (R2 - L2)
|
||||
|
||||
|
||||
def parse_footnoteLine(page_ID: int, page: fitz.Page, json_from_DocXchain_obj, exclude_bboxes):
|
||||
"""
|
||||
:param page_ID: int类型,当前page在当前pdf文档中是第page_D页。
|
||||
:param page :fitz读取的当前页的内容
|
||||
:param res_dir_path: str类型,是每一个pdf文档,在当前.py文件的目录下生成一个与pdf文档同名的文件夹,res_dir_path就是文件夹的dir
|
||||
:param json_from_DocXchain_obj: dict类型,把pdf文档送入DocXChain模型中后,提取bbox,结果保存到pdf文档同名文件夹下的 page_ID.json文件中了。json_from_DocXchain_obj就是打开后的dict
|
||||
"""
|
||||
DPI = 72 # use this resolution
|
||||
pix = page.get_pixmap(dpi=DPI)
|
||||
pageL = 0
|
||||
pageR = int(pix.w)
|
||||
pageU = 0
|
||||
pageD = int(pix.h)
|
||||
|
||||
#---------------------- PyMuPDF解析text --------------------#
|
||||
textSize_freq = collections.defaultdict(float) # text块中,textSize的频率
|
||||
textBlock_bboxs = []
|
||||
textLine_bboxs = []
|
||||
text_blocks = page.get_text(
|
||||
"dict",
|
||||
flags=fitz.TEXTFLAGS_TEXT,
|
||||
#clip=clip,
|
||||
)["blocks"]
|
||||
totText_list = []
|
||||
for i in range(len(text_blocks)):
|
||||
# print(blocks[i]) #### print
|
||||
bbox = text_blocks[i]['bbox']
|
||||
textBlock_bboxs.append(bbox)
|
||||
# print(bbox)
|
||||
cur_block_text_list = []
|
||||
for tt in text_blocks[i]['lines']:
|
||||
# 当前line
|
||||
cur_line_text_list = []
|
||||
cur_line_bbox = None # 当前line,最右侧的section的bbox
|
||||
for xf in tt['spans']:
|
||||
L, U, R, D = xf['bbox']
|
||||
L, R = min(L, R), max(L, R)
|
||||
U, D = min(U, D), max(U, D)
|
||||
textLine_bboxs.append((L, U, R, D))
|
||||
cur_line_text_list.append(xf['text'])
|
||||
textSize_freq[xf['size']] += len(xf['text'])
|
||||
cur_lines_text = ' '.join(cur_line_text_list)
|
||||
cur_block_text_list.append(cur_lines_text)
|
||||
totText_list.append('\n'.join(cur_block_text_list))
|
||||
totText = '\n'.join(totText_list)
|
||||
# print(totText) # 打印Text
|
||||
|
||||
textLine_bboxs.sort(key = lambda LURD: (LURD[0], LURD[1]))
|
||||
textBlock_bboxs.sort(key = lambda LURD: (LURD[0], LURD[1]))
|
||||
|
||||
# print('------------ textSize_freq -----------')
|
||||
max_sizeFreq = 0 # 出现频率最高的textSize
|
||||
textSize_withMaxFreq = 0
|
||||
for x, f in textSize_freq.items():
|
||||
# print(x, f)
|
||||
if f > max_sizeFreq:
|
||||
max_sizeFreq = f
|
||||
textSize_withMaxFreq = x
|
||||
#**********************************************************#
|
||||
|
||||
#------------------ PyMuPDF读取drawings -----------------#
|
||||
horizon_lines = []
|
||||
drawings = page.get_cdrawings()
|
||||
for drawing in drawings:
|
||||
try:
|
||||
rect = drawing['rect']
|
||||
L, U, R, D = rect
|
||||
# if (L, U, R, D) in exclude_bboxes:
|
||||
# continue # 如果是Fiugre, Table, Equation。注释掉是因为,可以暂时先不消,先自我对消。最后再判读需不需要排除。
|
||||
# 如果是水平线
|
||||
if U <= D and D - U <= 3:
|
||||
# 如果长度够
|
||||
if (pageR - pageL) / 15 <= R - L:
|
||||
if not(80/800 * pageD <= U <= 750/800 * pageD):
|
||||
continue # 很可能是页眉和页脚的线
|
||||
horizon_lines.append((L, U, R, D))
|
||||
# print((L, U, R, D))
|
||||
except:
|
||||
pass
|
||||
horizon_lines.sort(key = lambda LURD: (LURD[1]))
|
||||
#********************************************************#
|
||||
|
||||
#----------------- 两条线可能是在表格中 ------------------#
|
||||
def has_text_below_line(L: float, U: float, R: float, D: float, inLowerArea: bool) -> bool:
|
||||
"""
|
||||
检查线下是否紧挨着text
|
||||
"""
|
||||
Uu, Du = U - textSize_withMaxFreq, U # 线上的一个矩形
|
||||
Lu, Ru = L, R
|
||||
Ud, Dd = U, U + textSize_withMaxFreq # 线下的一个矩形
|
||||
Ld, Rd = L, R
|
||||
find = 0 # 在线下的文字。统计面积。
|
||||
leftTextCnt = 0 # 不在线底下的文字(整体在线左侧的文字),说明不是个脚注线。统计面积。
|
||||
English_alpha_cnt = 0 # 英文字母个数
|
||||
nonEnglish_alpha_cnt = 0 # 非英文字母个数
|
||||
punctuation_mark_cnt = 0 # 常见标点符号个数
|
||||
digit_cnt = 0 # 数字个数
|
||||
|
||||
distance_nearest_up_line = None
|
||||
distance_nearest_down_line = None
|
||||
|
||||
for i in range(len(text_blocks)):
|
||||
# print(blocks[i]) #### print
|
||||
bbox = text_blocks[i]['bbox']
|
||||
L0, U0, R0, D0 = bbox
|
||||
if 0< (R0 - L0) < pageR / 6 and (D0 - U0) / (R0 - L0) > 10 :
|
||||
continue # 一个很窄的,竖直的长条。比如,arXiv预印本,左侧的arXiv标志信息。
|
||||
textBlock_bboxs.append(bbox)
|
||||
# print(bbox)
|
||||
cur_block_text_list = []
|
||||
for tt in text_blocks[i]['lines']:
|
||||
# 当前line
|
||||
cur_line_text_list = []
|
||||
cur_line_bbox = None # 当前line,最右侧的section的bbox
|
||||
for xf in tt['spans']:
|
||||
L2, U2, R2, D2 = xf['bbox']
|
||||
L2, R2 = min(L2, R2), max(L2, R2)
|
||||
U2, D2 = min(U2, D2), max(U2, D2)
|
||||
textLine = xf['text']
|
||||
if L>0 and L2 < L and (L - L2) / L > 0.2:
|
||||
leftTextCnt += abs(R2 - L2) * abs(D2 - U2)
|
||||
else:
|
||||
## 线下的部分
|
||||
ratio_1, ratio_2 = calculate_overlapRatio_between_line1_and_line2(Ud, Dd, U2, D2)
|
||||
ratio_3, ratio_4 = calculate_overlapRatio_between_line1_and_line2(Ld, Rd, L2, R2)
|
||||
if U < (U2 + D2) / 2 and ratio_1 > 0 and ratio_2 > 0:
|
||||
if max(ratio_3, ratio_4) > 0.8:
|
||||
# if 444 <= U1 < 445 and 55 <= L2 < 56:
|
||||
# print('匹配的框', L2, U2, R2, D2)
|
||||
# if xf['size'] > 1.2 * textSize_withMaxFreq:
|
||||
# return False # 可能是个标题。不能这样卡
|
||||
find += abs(R2 - L2) * abs(D2 - U2)
|
||||
distance_nearest_down_line = (U2 + D2) / 2 - U
|
||||
for c in textLine:
|
||||
if c == ' ':
|
||||
continue
|
||||
elif c.isdigit() == True:
|
||||
digit_cnt += 1
|
||||
elif c in ',.:!?[]()%,。、!?:【】()《》-':
|
||||
punctuation_mark_cnt += 1
|
||||
elif c.isalpha() == True:
|
||||
English_alpha_cnt += 1
|
||||
else:
|
||||
nonEnglish_alpha_cnt += 1
|
||||
## 线上的部分
|
||||
ratio_5, ratio_6 = calculate_overlapRatio_between_line1_and_line2(Uu, Du, U2, D2)
|
||||
ratio_7, ratio_8 = calculate_overlapRatio_between_line1_and_line2(Lu, Ru, L2, R2)
|
||||
if (U2 + D2) / 2 < U and ratio_5 > 0 and ratio_6 > 0:
|
||||
if max(ratio_7, ratio_8) > 0.8:
|
||||
distance_nearest_up_line = U - (U2 + D2) / 2
|
||||
# if distance_nearest_up_line < 0:
|
||||
# print(Lu, Uu, Ru, Du, L2, U2, R2, D2)
|
||||
# print(distance_nearest_up_line, distance_nearest_down_line)
|
||||
if distance_nearest_up_line != None and distance_nearest_down_line != None:
|
||||
if distance_nearest_up_line * 1.5 < distance_nearest_down_line:
|
||||
return False # 如果,一根线。距离上面的文字line更近。说明是个下划线,而不是footnoteLine
|
||||
|
||||
## 在上面的线条,要考虑左侧的text块儿。在很靠下的线条,就暂时不考虑左侧text块儿了。
|
||||
if inLowerArea == False:
|
||||
if leftTextCnt >= 2000/500000 * pageR * pageD:
|
||||
return False
|
||||
return find >= 0 and (English_alpha_cnt + nonEnglish_alpha_cnt + digit_cnt) >= 10
|
||||
## 最下面区域的线条,判断时。
|
||||
# print(English_alpha_cnt, nonEnglish_alpha_cnt, digit_cnt)
|
||||
if (English_alpha_cnt + nonEnglish_alpha_cnt + digit_cnt) == 0:
|
||||
return False
|
||||
if (English_alpha_cnt + digit_cnt) / (English_alpha_cnt + nonEnglish_alpha_cnt + digit_cnt) > 0.5:
|
||||
if nonEnglish_alpha_cnt / (English_alpha_cnt + nonEnglish_alpha_cnt + digit_cnt) > 0.4:
|
||||
return False
|
||||
else:
|
||||
return True
|
||||
return True
|
||||
|
||||
|
||||
visited = [False for _ in range(len(horizon_lines))]
|
||||
for i, b1 in enumerate(horizon_lines):
|
||||
for j in range(i + 1, len(horizon_lines)):
|
||||
L1, U1, R1, D1 = horizon_lines[i]
|
||||
L2, U2, R2, D2 = horizon_lines[j]
|
||||
|
||||
## 在一条水平线,且挨着
|
||||
if L1 > L2:
|
||||
L1, U1, R1, D1, L2, U2, R2, D2 = L2, U2, R2, D2, L1, U1, R1, D1
|
||||
in_horizontal_line_flag = (max(U1, D1, U2, D2) - min(U1, D1, U2, D2) <= 5) and (L2 - R1 <= pageR/10)
|
||||
if in_horizontal_line_flag == True:
|
||||
visited[i] = True
|
||||
visited[j] = True
|
||||
|
||||
## 在竖直方向上是一致的。(表格,或者有的文章就是喜欢划线)
|
||||
L1, U1, R1, D1 = horizon_lines[i]
|
||||
L2, U2, R2, D2 = horizon_lines[j]
|
||||
ratio_1, ratio_2 = calculate_overlapRatio_between_line1_and_line2(L1, R1, L2, R2)
|
||||
# print(L1, U1, R1, D1, L2, U2, R2, D2, ratio_1, ratio_2)
|
||||
in_vertical_line_flag = (ratio_1 > 0.9 and ratio_2 > 0.9) or (max(ratio_1, ratio_2) > 0.95)
|
||||
if in_vertical_line_flag == True:
|
||||
visited[i] = True
|
||||
# if (U2 < pageD * 0.8 or (U2 - U1) < pageD * 0.3) and has_text_below_line(L2, U2, R2, D2, False) == False:
|
||||
# visited[j] = True # 最最底下的线先不要动
|
||||
else:
|
||||
if ratio_1 > 0 and (R2 - L2) / (R1 - L1) > 1:
|
||||
visited[i] = True
|
||||
# print(horizon_lines)
|
||||
horizon_lines = [horizon_lines[i] for i in range(len(horizon_lines)) if visited[i] == False]
|
||||
# print(horizon_lines)
|
||||
#*****************************************************************#
|
||||
|
||||
#------- 靠上的,就不是脚注。用一个THRESHOLD直接卡掉位于上半页的 -------#
|
||||
visited = [False for _ in range(len(horizon_lines))]
|
||||
THRESHOLD = (pageD - pageU) * 0.5
|
||||
for i, (L, U, R, D) in enumerate(horizon_lines):
|
||||
if U < THRESHOLD:
|
||||
visited[i] = True
|
||||
horizon_lines = [horizon_lines[i] for i in range(len(horizon_lines)) if visited[i] == False]
|
||||
#******************************************************#
|
||||
|
||||
#--------------- 此时,还有遮挡的,上面的丢弃 ---------------#
|
||||
visited = [False for _ in range(len(horizon_lines))]
|
||||
for i, (L1, U1, R1, D1) in enumerate(horizon_lines):
|
||||
for j in range(i + 1, len(horizon_lines)):
|
||||
L2, U2, R2, D2 = horizon_lines[j]
|
||||
ratio_1, ratio_2 = calculate_overlapRatio_between_line1_and_line2(L1, R1, L2, R2)
|
||||
if (ratio_1 > 0.2 and ratio_2 > 0.2) or max(ratio_1, ratio_2) > 0.7:
|
||||
visited[i] = True
|
||||
horizon_lines = [horizon_lines[i] for i in range(len(horizon_lines)) if visited[i] == False]
|
||||
#********************************************************#
|
||||
# print(horizon_lines)
|
||||
## 检查,线下面有没有紧挨着的text
|
||||
horizon_lines = [LURD for LURD in horizon_lines if has_text_below_line(*(LURD), True) == True]
|
||||
# print(horizon_lines)
|
||||
## 卡一下长度
|
||||
# horizon_lines = [LURD for LURD in horizon_lines if (LURD[2] - LURD[0] >= pageR / 10)]
|
||||
|
||||
## 上面最多保留2条
|
||||
horizon_lines = horizon_lines[max(-2, -len(horizon_lines)) :]
|
||||
|
||||
|
||||
#----------------------------------------------------- 第2段 -----------------------------------------------------------#
|
||||
#----------------------------------- 最下面的情形,用距离硬卡。还有在右侧的情形就被包含了 -----------------------------------#
|
||||
#------------------ PyMuPDF读取drawings -----------------#
|
||||
down_horizon_lines = []
|
||||
|
||||
drawings = page.get_cdrawings()
|
||||
for drawing in drawings:
|
||||
try:
|
||||
rect = drawing['rect']
|
||||
L, U, R, D = rect
|
||||
# if (L, U, R, D) in exclude_bboxes:
|
||||
# continue # 如果是Fiugre, Table, Equation。目前是Figure识别的比较好。但是Table和Equation识别的不好
|
||||
# 如果是水平线
|
||||
if U <= D and D - U <= 3 and U > pageD * 0.85:
|
||||
# 如果长度够
|
||||
if (pageR - pageL) / 15 <= R - L:
|
||||
down_horizon_lines.append((L, U, R, D))
|
||||
# print((L, U, R, D))
|
||||
except:
|
||||
pass
|
||||
|
||||
down_horizon_lines.sort(key = lambda LURD: (LURD[0], LURD[2], LURD[1]))
|
||||
visited = [False for _ in range(len(down_horizon_lines))]
|
||||
for i in range(len(down_horizon_lines) - 1):
|
||||
L1, U1, R1, D1 = down_horizon_lines[i]
|
||||
L2, U2, R2, D2 = down_horizon_lines[i + 1]
|
||||
ratio_1, ratio_2 = calculate_overlapRatio_between_line1_and_line2(L1, R1, L2, R2)
|
||||
if ratio_1 <= 0.1 and ratio_2 <= 0.1:
|
||||
if L2 - R1 <= pageR / 3:
|
||||
visited[i] = True
|
||||
visited[i + 1] = True
|
||||
down_horizon_lines = [down_horizon_lines[i] for i in range(len(down_horizon_lines)) if visited[i] == False]
|
||||
|
||||
down_horizon_lines = [LURD for LURD in down_horizon_lines if has_text_below_line(*(LURD), True) == True]
|
||||
# for LURD in down_horizon_lines:
|
||||
# print('第2阶段,LURD是: ', LURD)
|
||||
# print(has_text_below_line(*(LURD), True))
|
||||
|
||||
footnoteLines = horizon_lines + down_horizon_lines
|
||||
footnoteLines = list(set(footnoteLines))
|
||||
footnoteLines = footnoteLines[max(-2, -len(footnoteLines)) : ]
|
||||
|
||||
#-------------------------- 最后再检查一遍。是否在图片、表格、公式中。 ------------------------------#
|
||||
def line_in_specialBboxes(L: float, U: float, R: float, D: float, specialBboxes) -> bool:
|
||||
L2, U2, R2, D2 = L, U, R, D # 当前这根线
|
||||
for L1, U1, R1, D1 in specialBboxes:
|
||||
if U1 <= U2 <= D2 < D1:
|
||||
ratio_1, ratio_2 = calculate_overlapRatio_between_line1_and_line2(L1, R1, L2, R2)
|
||||
if ratio_1 > 0 and ratio_2 > 0.6:
|
||||
return True
|
||||
# else:
|
||||
# U1 -= min(textSize_withMaxFreq * 2, 20)
|
||||
# D1 += min(textSize_withMaxFreq * 2, 20)
|
||||
# if U1 <= U2 <= D2 < D1:
|
||||
# ratio_1, ratio_2 = calculate_overlapRatio_between_line1_and_line2(L1, R1, L2, R2)
|
||||
# if ratio_1 > 0 and ratio_2 > 0.8:
|
||||
# return True
|
||||
return False
|
||||
|
||||
footnoteLines = [LURD for LURD in footnoteLines if line_in_specialBboxes(*(LURD), exclude_bboxes) == False]
|
||||
|
||||
#-------------------------- 检查,线,是否在当前column的左侧,而不是在一段文字的中间 (通过DocXChain识别的column或者徐超老师写的Layout识别)------------------------------#
|
||||
# #--------- 通过json_from_DocXchain来获取 column ---------#
|
||||
# column_bbox_from_DocXChain = []
|
||||
|
||||
# xf_json = json_from_DocXchain_obj
|
||||
# width_from_json = xf_json['page_info']['width']
|
||||
# height_from_json = xf_json['page_info']['height']
|
||||
# LR_scaleRatio = width_from_json / (pageR - pageL)
|
||||
# UD_scaleRatio = height_from_json / (pageD - pageU)
|
||||
|
||||
# # {0: 'title', # 标题
|
||||
# # 1: 'figure', # 图片
|
||||
# # 2: 'plain text', # 文本
|
||||
# # 3: 'header', # 页眉
|
||||
# # 4: 'page number', # 页码
|
||||
# # 5: 'footnote', # 脚注
|
||||
# # 6: 'footer', # 页脚
|
||||
# # 7: 'table', # 表格
|
||||
# # 8: 'table caption', # 表格描述
|
||||
# # 9: 'figure caption', # 图片描述
|
||||
# # 10: 'equation', # 公式
|
||||
# # 11: 'full column', # 单栏
|
||||
# # 12: 'sub column', # 多栏
|
||||
# # 13: 'embedding', # 嵌入公式
|
||||
# # 14: 'isolated'} # 单行公式
|
||||
# for xf in xf_json['layout_dets']:
|
||||
# L = xf['poly'][0] / LR_scaleRatio
|
||||
# U = xf['poly'][1] / UD_scaleRatio
|
||||
# R = xf['poly'][2] / LR_scaleRatio
|
||||
# D = xf['poly'][5] / UD_scaleRatio
|
||||
# # L += pageL # 有的页面,artBox偏移了。不在(0,0)
|
||||
# # R += pageL
|
||||
# # U += pageU
|
||||
# # D += pageU
|
||||
# L, R = min(L, R), max(L, R)
|
||||
# U, D = min(U, D), max(U, D)
|
||||
# if (xf['category_id'] == 11 or xf['category_id'] == 12) and xf['score'] >= 0.3:
|
||||
# column_bbox_from_DocXChain.append((L, U, R, D))
|
||||
|
||||
#---------------手写,检查,线是否是与某个column的左端对齐 ------------------#
|
||||
def check_isOnTheLeftOfColumn(L: float, U: float, R: float, D: float) -> bool:
|
||||
LL = L - textSize_withMaxFreq
|
||||
RR = LL
|
||||
UU = max(pageD * 0.02, U - 100/800 * pageD)
|
||||
DD = min(U + 50/800 * pageD, pageD * 0.98)
|
||||
|
||||
# print(LL, UU, RR, DD)
|
||||
cnt = 0
|
||||
for bbox in textLine_bboxs:
|
||||
L2, U2, R2, D2 = bbox
|
||||
ratio_1, ratio_2 = calculate_overlapRatio_between_line1_and_line2(UU, DD, U2, D2)
|
||||
ratio_3, ratio_4 = calculate_overlapRatio_between_line1_and_line2(L, R, L2, R2)
|
||||
if ratio_1 > 0 and ratio_2 > 0:
|
||||
if max(ratio_3, ratio_4) > 0.8:
|
||||
if abs(LL - L2) <= 20/700 * pageR:
|
||||
cnt += 1
|
||||
# else:
|
||||
# if (R2 - L2) >= 30/700 * pageR:
|
||||
# print(LL, UU, RR, DD, L2, U2, R2, D2)
|
||||
# return False # 不能这样卡。有些注释里面,单独的特殊符号就是一个textLineBbox
|
||||
# print('cnt: ', cnt)
|
||||
return cnt >= 4
|
||||
|
||||
# def check_isOnTheLeftOfColumn_considerLayout(L0: float, U0: float, R0: float, D0: float) -> bool:
|
||||
# LL = L0 - textSize_withMaxFreq * 1.5
|
||||
# RR = LL
|
||||
# UU = 100/800 * pageD
|
||||
# DD = 700/800 * pageD
|
||||
|
||||
# STEP = textSize_withMaxFreq / 2
|
||||
|
||||
# def check_ok(L: float, U: float, R: float, D: float) -> bool:
|
||||
# for bbox in textBlock_bboxs:
|
||||
# L2, U2, R2, D2 = bbox
|
||||
# ratio_3, ratio_4 = calculate_overlapRatio_between_line1_and_line2(L, R, L2, R2)
|
||||
# if max(ratio_3, ratio_4) > 0.8:
|
||||
# if (R2 - L2) > 1/4 * pageR and L2 < LL <= RR < R2:
|
||||
# if abs(LL - L2) < 50/700 * pageR or abs(RR - R2) < 50/700 * pageR:
|
||||
# continue
|
||||
# else:
|
||||
# return False
|
||||
# return True
|
||||
|
||||
# ## 先探上面
|
||||
# u = UU
|
||||
# d = U0
|
||||
# while u + STEP/2 < d:
|
||||
# mid = (u + d) / 2
|
||||
# if check_ok(L0, mid, R0, U0) == True:
|
||||
# d = mid
|
||||
# else:
|
||||
# u = mid + STEP
|
||||
# print(mid)
|
||||
# dist_up = U0 - u
|
||||
# print(u)
|
||||
# ## 再探下面
|
||||
# u = D0
|
||||
# d = DD
|
||||
# while u + STEP/2 < d:
|
||||
# mid = (u + d) / 2
|
||||
# if check_ok(L0, mid, R0, D0) == True:
|
||||
# u = mid
|
||||
# else:
|
||||
# d = mid - STEP
|
||||
# print(u)
|
||||
# print('^^^^^^^^^^^^^^')
|
||||
# dist_down = u - D0
|
||||
|
||||
# if dist_up + dist_down < textSize_withMaxFreq * 10:
|
||||
# return False
|
||||
# return True
|
||||
|
||||
|
||||
footnoteLines = [LURD for LURD in footnoteLines if check_isOnTheLeftOfColumn(*(LURD)) == True]
|
||||
# footnoteLines = [LURD for LURD in footnoteLines if check_isOnTheLeftOfColumn_considerLayout(*(LURD)) == True] # 不具有泛化性。不用了。
|
||||
|
||||
#--------------------------------- 通过footnoteLine获取bbox -------------------------------#
|
||||
def get_footnoteBbox(L: float, U: float, R: float, D: float) -> (float, float, float, float):
|
||||
"""
|
||||
检查线下是否紧挨着text
|
||||
"""
|
||||
L1, U1, R1, D1 = L, U, R, D
|
||||
raw_bboxes = []
|
||||
for i in range(len(text_blocks)):
|
||||
bbox = text_blocks[i]['bbox']
|
||||
L2, U2, R2, D2 = bbox
|
||||
if (D2 - U2) / (R2 - L2) > 10 and (R2 - L2) < pageR / 6:
|
||||
continue # 一个很窄的,竖直的长条。比如,arXiv预印本,左侧的arXiv标志信息。
|
||||
if U2 < D2 < U1:
|
||||
continue # 在线上面
|
||||
under_THRESHOLD = min(D1 + textSize_withMaxFreq * 20, pageD * 0.98)
|
||||
if U2 < under_THRESHOLD:
|
||||
ratio_1, ratio_2 = calculate_overlapRatio_between_line1_and_line2(L1, R1, L2, R2)
|
||||
if max(ratio_1, ratio_2) > 0.8:
|
||||
raw_bboxes.append((L2, U2, R2, D2))
|
||||
# print(L1, U1, R1, D1)
|
||||
# print(raw_bboxes)
|
||||
if len(raw_bboxes) == 0:
|
||||
return []
|
||||
|
||||
raw_bboxes.sort(key = lambda LURD: (LURD[1], LURD[0]))
|
||||
raw_bboxes = [LURD for LURD in raw_bboxes if (abs(LURD[0] - L1) < textSize_withMaxFreq * 6 or L1 < LURD[0])] # footnote的bbox,应该都是左端对齐的
|
||||
if len(raw_bboxes) == 0:
|
||||
return []
|
||||
#------------------ full column和sub column混合,肯定也不行 ------------------#
|
||||
LL, UU, RR, DD = raw_bboxes[0]
|
||||
for L, U, R, D in raw_bboxes:
|
||||
LL, UU, RR, DD = min(LL, L), min(UU, U), max(RR, R), max(DD, D)
|
||||
for L, U, R, D in raw_bboxes:
|
||||
if (RR - LL) > pageR*0.8 and (R - L) > pageR * 0.15 and (RR - LL) / (R - L) > 2:
|
||||
return []
|
||||
if abs(LL - L) > textSize_withMaxFreq * 3:
|
||||
return []
|
||||
|
||||
#-------------------- 太高了的,full column的框。不行 ----------------------#
|
||||
if UU < 650/800 * pageD and (RR - LL) > 0.5 * pageR:
|
||||
return []
|
||||
|
||||
#-------------- 第一段字数很少。后面的段字数很多,也不行 ----------------#
|
||||
if len(raw_bboxes) > 1:
|
||||
bbox_square = []
|
||||
for L, U, R, D in raw_bboxes:
|
||||
cur_s = abs(R - L) * abs(D - U)
|
||||
bbox_square.append(cur_s)
|
||||
|
||||
s0 = bbox_square[0]
|
||||
s1n = sum(bbox_square[1: ]) / len(bbox_square[1: ])
|
||||
if s1n / s0 > 10 or max(bbox_square) / s0 > 15:
|
||||
return []
|
||||
|
||||
raw_bboxes += [(LL, UU, RR, DD)]
|
||||
return raw_bboxes
|
||||
|
||||
# print(footnoteLines)
|
||||
footnoteBboxes = []
|
||||
for L, U, R, D in footnoteLines:
|
||||
cur = get_footnoteBbox(L, U, R, D)
|
||||
if len(cur) > 0:
|
||||
footnoteBboxes.append((L, U, R, D))
|
||||
footnoteBboxes += cur
|
||||
|
||||
footnoteBboxes = list(set(footnoteBboxes))
|
||||
return footnoteBboxes
|
||||
|
||||
|
||||
def __bbox_in(box1, box2):
|
||||
"""
|
||||
box1是否在box2中
|
||||
"""
|
||||
L1, U1, R1, D1 = box1
|
||||
L2, U2, R2, D2 = box2
|
||||
if int(L2) <= int(L1) and int(U2) <= int(U1) and int(R1) <= int(R2) and int(D1) <= int(D2):
|
||||
return True
|
||||
return False
|
||||
|
||||
def remove_footnote_text(raw_text_block, footnote_bboxes):
|
||||
"""
|
||||
:param raw_text_block: str类型,是当前页的文本内容
|
||||
:param footnoteBboxes: list类型,是当前页的脚注bbox
|
||||
"""
|
||||
footnote_text_blocks = []
|
||||
for block in raw_text_block:
|
||||
text_bbox = block['bbox']
|
||||
# TODO 更严谨点在line级别做
|
||||
if any([_is_in_or_part_overlap(text_bbox, footnote_bbox) for footnote_bbox in footnote_bboxes]):
|
||||
#if any([text_bbox[3]>=footnote_bbox[1] for footnote_bbox in footnote_bboxes]):
|
||||
block['tag'] = 'footnote'
|
||||
footnote_text_blocks.append(block)
|
||||
#raw_text_block.remove(block)
|
||||
|
||||
# 移除,不能再内部移除,否则会出错
|
||||
for block in footnote_text_blocks:
|
||||
raw_text_block.remove(block)
|
||||
|
||||
return raw_text_block, footnote_text_blocks
|
||||
|
||||
def remove_footnote_image(image_blocks, footnote_bboxes):
|
||||
"""
|
||||
:param image_bboxes: list类型,是当前页的图片bbox(结构体)
|
||||
:param footnoteBboxes: list类型,是当前页的脚注bbox
|
||||
"""
|
||||
footnote_imgs_blocks = []
|
||||
for image_block in image_blocks:
|
||||
if any([__bbox_in(image_block['bbox'], footnote_bbox) for footnote_bbox in footnote_bboxes]):
|
||||
footnote_imgs_blocks.append(image_block)
|
||||
|
||||
for footnote_imgs_block in footnote_imgs_blocks:
|
||||
image_blocks.remove(footnote_imgs_block)
|
||||
|
||||
return image_blocks, footnote_imgs_blocks
|
||||
|
||||
|
||||
def remove_headder_footer_one_page(text_raw_blocks, image_bboxes, table_bboxes, header_bboxs, footer_bboxs, page_no_bboxs, page_w, page_h):
|
||||
"""
|
||||
删除页眉页脚,页码
|
||||
从line级别进行删除,删除之后观察这个text-block是否是空的,如果是空的,则移动到remove_list中
|
||||
"""
|
||||
header = []
|
||||
footer = []
|
||||
if len(header)==0:
|
||||
model_header = header_bboxs
|
||||
if model_header:
|
||||
x0 = min([x for x,_,_,_ in model_header])
|
||||
y0 = min([y for _,y,_,_ in model_header])
|
||||
x1 = max([x1 for _,_,x1,_ in model_header])
|
||||
y1 = max([y1 for _,_,_,y1 in model_header])
|
||||
header = [x0, y0, x1, y1]
|
||||
if len(footer)==0:
|
||||
model_footer = footer_bboxs
|
||||
if model_footer:
|
||||
x0 = min([x for x,_,_,_ in model_footer])
|
||||
y0 = min([y for _,y,_,_ in model_footer])
|
||||
x1 = max([x1 for _,_,x1,_ in model_footer])
|
||||
y1 = max([y1 for _,_,_,y1 in model_footer])
|
||||
footer = [x0, y0, x1, y1]
|
||||
|
||||
|
||||
header_y0 = 0 if len(header) == 0 else header[3]
|
||||
footer_y0 = page_h if len(footer) == 0 else footer[1]
|
||||
if page_no_bboxs:
|
||||
top_part = [b for b in page_no_bboxs if b[3] < page_h/2]
|
||||
btn_part = [b for b in page_no_bboxs if b[1] > page_h/2]
|
||||
|
||||
top_max_y0 = max([b[1] for b in top_part]) if top_part else 0
|
||||
btn_min_y1 = min([b[3] for b in btn_part]) if btn_part else page_h
|
||||
|
||||
header_y0 = max(header_y0, top_max_y0)
|
||||
footer_y0 = min(footer_y0, btn_min_y1)
|
||||
|
||||
content_boundry = [0, header_y0, page_w, footer_y0]
|
||||
|
||||
header = [0,0, page_w, header_y0]
|
||||
footer = [0, footer_y0, page_w, page_h]
|
||||
|
||||
"""以上计算出来了页眉页脚的边界,下面开始进行删除"""
|
||||
text_block_to_remove = []
|
||||
# 首先检查每个textblock
|
||||
for blk in text_raw_blocks:
|
||||
if len(blk['lines']) > 0:
|
||||
for line in blk['lines']:
|
||||
line_del = []
|
||||
for span in line['spans']:
|
||||
span_del = []
|
||||
if span['bbox'][3] < header_y0:
|
||||
span_del.append(span)
|
||||
elif _is_in_or_part_overlap(span['bbox'], header) or _is_in_or_part_overlap(span['bbox'], footer):
|
||||
span_del.append(span)
|
||||
for span in span_del:
|
||||
line['spans'].remove(span)
|
||||
if not line['spans']:
|
||||
line_del.append(line)
|
||||
|
||||
for line in line_del:
|
||||
blk['lines'].remove(line)
|
||||
else:
|
||||
# if not blk['lines']:
|
||||
blk['tag'] = 'in-foot-header-area'
|
||||
text_block_to_remove.append(blk)
|
||||
|
||||
"""有的时候由于pageNo太小了,总是会有一点和content_boundry重叠一点,被放入正文,因此对于pageNo,进行span粒度的删除"""
|
||||
page_no_block_2_remove = []
|
||||
if page_no_bboxs:
|
||||
for pagenobox in page_no_bboxs:
|
||||
for block in text_raw_blocks:
|
||||
if _is_in_or_part_overlap(pagenobox, block['bbox']): # 在span级别删除页码
|
||||
for line in block['lines']:
|
||||
for span in line['spans']:
|
||||
if _is_in_or_part_overlap(pagenobox, span['bbox']):
|
||||
#span['text'] = ''
|
||||
span['tag'] = "page-no"
|
||||
# 检查这个block是否只有这一个span,如果是,那么就把这个block也删除
|
||||
if len(line['spans']) == 1 and len(block['lines'])==1:
|
||||
page_no_block_2_remove.append(block)
|
||||
else:
|
||||
# 测试最后一个是不是页码:规则是,最后一个block仅有1个line,一个span,且text是数字,空格,符号组成,不含字母,并且包含数字
|
||||
if len(text_raw_blocks) > 0:
|
||||
text_raw_blocks.sort(key=lambda x: x['bbox'][1], reverse=True)
|
||||
last_block = text_raw_blocks[0]
|
||||
if len(last_block['lines']) == 1:
|
||||
last_line = last_block['lines'][0]
|
||||
if len(last_line['spans']) == 1:
|
||||
last_span = last_line['spans'][0]
|
||||
if last_span['text'].strip() and not re.search('[a-zA-Z]', last_span['text']) and re.search('[0-9]', last_span['text']):
|
||||
last_span['tag'] = "page-no"
|
||||
page_no_block_2_remove.append(last_block)
|
||||
|
||||
|
||||
for b in page_no_block_2_remove:
|
||||
text_block_to_remove.append(b)
|
||||
|
||||
for blk in text_block_to_remove:
|
||||
if blk in text_raw_blocks:
|
||||
text_raw_blocks.remove(blk)
|
||||
|
||||
text_block_remain = text_raw_blocks
|
||||
image_bbox_to_remove = [bbox for bbox in image_bboxes if not _is_in_or_part_overlap(bbox, content_boundry)]
|
||||
|
||||
image_bbox_remain = [bbox for bbox in image_bboxes if _is_in_or_part_overlap(bbox, content_boundry)]
|
||||
table_bbox_to_remove = [bbox for bbox in table_bboxes if not _is_in_or_part_overlap(bbox, content_boundry)]
|
||||
table_bbox_remain = [bbox for bbox in table_bboxes if _is_in_or_part_overlap(bbox, content_boundry)]
|
||||
|
||||
return image_bbox_remain, table_bbox_remain, text_block_remain, text_block_to_remove, image_bbox_to_remove, table_bbox_to_remove
|
||||
@@ -0,0 +1,77 @@
|
||||
import os
|
||||
import collections # 统计库
|
||||
import re # 正则
|
||||
from libs.commons import fitz # pyMuPDF库
|
||||
import json # json
|
||||
|
||||
|
||||
def parse_headers(page_ID: int, page: fitz.Page, json_from_DocXchain_obj: dict):
|
||||
"""
|
||||
:param page_ID: int类型,当前page在当前pdf文档中是第page_D页。
|
||||
:param page :fitz读取的当前页的内容
|
||||
:param res_dir_path: str类型,是每一个pdf文档,在当前.py文件的目录下生成一个与pdf文档同名的文件夹,res_dir_path就是文件夹的dir
|
||||
:param json_from_DocXchain_obj: dict类型,把pdf文档送入DocXChain模型中后,提取bbox,结果保存到pdf文档同名文件夹下的 page_ID.json文件中了。json_from_DocXchain_obj就是打开后的dict
|
||||
"""
|
||||
DPI = 72 # use this resolution
|
||||
pix = page.get_pixmap(dpi=DPI)
|
||||
pageL = 0
|
||||
pageR = int(pix.w)
|
||||
pageU = 0
|
||||
pageD = int(pix.h)
|
||||
|
||||
|
||||
#--------- 通过json_from_DocXchain来获取 header ---------#
|
||||
header_bbox_from_DocXChain = []
|
||||
|
||||
xf_json = json_from_DocXchain_obj
|
||||
width_from_json = xf_json['page_info']['width']
|
||||
height_from_json = xf_json['page_info']['height']
|
||||
LR_scaleRatio = width_from_json / (pageR - pageL)
|
||||
UD_scaleRatio = height_from_json / (pageD - pageU)
|
||||
|
||||
# {0: 'title', # 标题
|
||||
# 1: 'figure', # 图片
|
||||
# 2: 'plain text', # 文本
|
||||
# 3: 'header', # 页眉
|
||||
# 4: 'page number', # 页码
|
||||
# 5: 'footnote', # 脚注
|
||||
# 6: 'footer', # 页脚
|
||||
# 7: 'table', # 表格
|
||||
# 8: 'table caption', # 表格描述
|
||||
# 9: 'figure caption', # 图片描述
|
||||
# 10: 'equation', # 公式
|
||||
# 11: 'full column', # 单栏
|
||||
# 12: 'sub column', # 多栏
|
||||
# 13: 'embedding', # 嵌入公式
|
||||
# 14: 'isolated'} # 单行公式
|
||||
for xf in xf_json['layout_dets']:
|
||||
L = xf['poly'][0] / LR_scaleRatio
|
||||
U = xf['poly'][1] / UD_scaleRatio
|
||||
R = xf['poly'][2] / LR_scaleRatio
|
||||
D = xf['poly'][5] / UD_scaleRatio
|
||||
# L += pageL # 有的页面,artBox偏移了。不在(0,0)
|
||||
# R += pageL
|
||||
# U += pageU
|
||||
# D += pageU
|
||||
L, R = min(L, R), max(L, R)
|
||||
U, D = min(U, D), max(U, D)
|
||||
if xf['category_id'] == 3 and xf['score'] >= 0.3:
|
||||
header_bbox_from_DocXChain.append((L, U, R, D))
|
||||
|
||||
|
||||
header_final_names = []
|
||||
header_final_bboxs = []
|
||||
header_ID = 0
|
||||
for L, U, R, D in header_bbox_from_DocXChain:
|
||||
# cur_header = page.get_pixmap(clip=(L,U,R,D))
|
||||
new_header_name = "header_{}_{}.png".format(page_ID, header_ID) # 页眉name
|
||||
# cur_header.save(res_dir_path + '/' + new_header_name) # 把页眉存储在新建的文件夹,并命名
|
||||
header_final_names.append(new_header_name) # 把页面的名字存在list中
|
||||
header_final_bboxs.append((L, U, R, D))
|
||||
header_ID += 1
|
||||
|
||||
|
||||
header_final_bboxs.sort(key = lambda LURD: (LURD[1], LURD[0]))
|
||||
curPage_all_header_bboxs = header_final_bboxs
|
||||
return curPage_all_header_bboxs
|
||||
|
||||
@@ -0,0 +1,77 @@
|
||||
import os
|
||||
import collections # 统计库
|
||||
import re # 正则
|
||||
from libs.commons import fitz # pyMuPDF库
|
||||
import json # json
|
||||
|
||||
|
||||
def parse_pageNos(page_ID: int, page: fitz.Page, json_from_DocXchain_obj: dict):
|
||||
"""
|
||||
:param page_ID: int类型,当前page在当前pdf文档中是第page_D页。
|
||||
:param page :fitz读取的当前页的内容
|
||||
:param res_dir_path: str类型,是每一个pdf文档,在当前.py文件的目录下生成一个与pdf文档同名的文件夹,res_dir_path就是文件夹的dir
|
||||
:param json_from_DocXchain_obj: dict类型,把pdf文档送入DocXChain模型中后,提取bbox,结果保存到pdf文档同名文件夹下的 page_ID.json文件中了。json_from_DocXchain_obj就是打开后的dict
|
||||
"""
|
||||
DPI = 72 # use this resolution
|
||||
pix = page.get_pixmap(dpi=DPI)
|
||||
pageL = 0
|
||||
pageR = int(pix.w)
|
||||
pageU = 0
|
||||
pageD = int(pix.h)
|
||||
|
||||
|
||||
#--------- 通过json_from_DocXchain来获取 pageNo ---------#
|
||||
pageNo_bbox_from_DocXChain = []
|
||||
|
||||
xf_json = json_from_DocXchain_obj
|
||||
width_from_json = xf_json['page_info']['width']
|
||||
height_from_json = xf_json['page_info']['height']
|
||||
LR_scaleRatio = width_from_json / (pageR - pageL)
|
||||
UD_scaleRatio = height_from_json / (pageD - pageU)
|
||||
|
||||
# {0: 'title', # 标题
|
||||
# 1: 'figure', # 图片
|
||||
# 2: 'plain text', # 文本
|
||||
# 3: 'header', # 页眉
|
||||
# 4: 'page number', # 页码
|
||||
# 5: 'footnote', # 脚注
|
||||
# 6: 'footer', # 页脚
|
||||
# 7: 'table', # 表格
|
||||
# 8: 'table caption', # 表格描述
|
||||
# 9: 'figure caption', # 图片描述
|
||||
# 10: 'equation', # 公式
|
||||
# 11: 'full column', # 单栏
|
||||
# 12: 'sub column', # 多栏
|
||||
# 13: 'embedding', # 嵌入公式
|
||||
# 14: 'isolated'} # 单行公式
|
||||
for xf in xf_json['layout_dets']:
|
||||
L = xf['poly'][0] / LR_scaleRatio
|
||||
U = xf['poly'][1] / UD_scaleRatio
|
||||
R = xf['poly'][2] / LR_scaleRatio
|
||||
D = xf['poly'][5] / UD_scaleRatio
|
||||
# L += pageL # 有的页面,artBox偏移了。不在(0,0)
|
||||
# R += pageL
|
||||
# U += pageU
|
||||
# D += pageU
|
||||
L, R = min(L, R), max(L, R)
|
||||
U, D = min(U, D), max(U, D)
|
||||
if xf['category_id'] == 4 and xf['score'] >= 0.3:
|
||||
pageNo_bbox_from_DocXChain.append((L, U, R, D))
|
||||
|
||||
|
||||
pageNo_final_names = []
|
||||
pageNo_final_bboxs = []
|
||||
pageNo_ID = 0
|
||||
for L, U, R, D in pageNo_bbox_from_DocXChain:
|
||||
# cur_pageNo = page.get_pixmap(clip=(L,U,R,D))
|
||||
new_pageNo_name = "pageNo_{}_{}.png".format(page_ID, pageNo_ID) # 页码name
|
||||
# cur_pageNo.save(res_dir_path + '/' + new_pageNo_name) # 把页码存储在新建的文件夹,并命名
|
||||
pageNo_final_names.append(new_pageNo_name) # 把页码的名字存在list中
|
||||
pageNo_final_bboxs.append((L, U, R, D))
|
||||
pageNo_ID += 1
|
||||
|
||||
|
||||
pageNo_final_bboxs.sort(key = lambda LURD: (LURD[1], LURD[0]))
|
||||
curPage_all_pageNo_bboxs = pageNo_final_bboxs
|
||||
return curPage_all_pageNo_bboxs
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,66 @@
|
||||
import os
|
||||
import collections # 统计库
|
||||
import re # 正则
|
||||
from libs.commons import fitz # pyMuPDF库
|
||||
import json # json
|
||||
|
||||
|
||||
def parse_tables(page_ID: int, page: fitz.Page, json_from_DocXchain_obj: dict):
|
||||
"""
|
||||
:param page_ID: int类型,当前page在当前pdf文档中是第page_D页。
|
||||
:param page :fitz读取的当前页的内容
|
||||
:param res_dir_path: str类型,是每一个pdf文档,在当前.py文件的目录下生成一个与pdf文档同名的文件夹,res_dir_path就是文件夹的dir
|
||||
:param json_from_DocXchain_obj: dict类型,把pdf文档送入DocXChain模型中后,提取bbox,结果保存到pdf文档同名文件夹下的 page_ID.json文件中了。json_from_DocXchain_obj就是打开后的dict
|
||||
"""
|
||||
DPI = 72 # use this resolution
|
||||
pix = page.get_pixmap(dpi=DPI)
|
||||
pageL = 0
|
||||
pageR = int(pix.w)
|
||||
pageU = 0
|
||||
pageD = int(pix.h)
|
||||
|
||||
|
||||
#--------- 通过json_from_DocXchain来获取 table ---------#
|
||||
table_bbox_from_DocXChain = []
|
||||
|
||||
xf_json = json_from_DocXchain_obj
|
||||
width_from_json = xf_json['page_info']['width']
|
||||
height_from_json = xf_json['page_info']['height']
|
||||
LR_scaleRatio = width_from_json / (pageR - pageL)
|
||||
UD_scaleRatio = height_from_json / (pageD - pageU)
|
||||
|
||||
|
||||
for xf in xf_json['layout_dets']:
|
||||
# {0: 'title', 1: 'figure', 2: 'plain text', 3: 'header', 4: 'page number', 5: 'footnote', 6: 'footer', 7: 'table', 8: 'table caption', 9: 'figure caption', 10: 'equation', 11: 'full column', 12: 'sub column'}
|
||||
# 13: 'embedding', # 嵌入公式
|
||||
# 14: 'isolated'} # 单行公式
|
||||
L = xf['poly'][0] / LR_scaleRatio
|
||||
U = xf['poly'][1] / UD_scaleRatio
|
||||
R = xf['poly'][2] / LR_scaleRatio
|
||||
D = xf['poly'][5] / UD_scaleRatio
|
||||
# L += pageL # 有的页面,artBox偏移了。不在(0,0)
|
||||
# R += pageL
|
||||
# U += pageU
|
||||
# D += pageU
|
||||
L, R = min(L, R), max(L, R)
|
||||
U, D = min(U, D), max(U, D)
|
||||
if xf['category_id'] == 7 and xf['score'] >= 0.3:
|
||||
table_bbox_from_DocXChain.append((L, U, R, D))
|
||||
|
||||
|
||||
table_final_names = []
|
||||
table_final_bboxs = []
|
||||
table_ID = 0
|
||||
for L, U, R, D in table_bbox_from_DocXChain:
|
||||
# cur_table = page.get_pixmap(clip=(L,U,R,D))
|
||||
new_table_name = "table_{}_{}.png".format(page_ID, table_ID) # 表格name
|
||||
# cur_table.save(res_dir_path + '/' + new_table_name) # 把表格存出在新建的文件夹,并命名
|
||||
table_final_names.append(new_table_name) # 把表格的名字存在list中,方便在md中插入引用
|
||||
table_final_bboxs.append((L, U, R, D))
|
||||
table_ID += 1
|
||||
|
||||
|
||||
table_final_bboxs.sort(key = lambda LURD: (LURD[1], LURD[0]))
|
||||
curPage_all_table_bboxs = table_final_bboxs
|
||||
return curPage_all_table_bboxs
|
||||
|
||||
@@ -0,0 +1,77 @@
|
||||
import os
|
||||
import collections # 统计库
|
||||
import re # 正则
|
||||
from libs.commons import fitz # pyMuPDF库
|
||||
import json # json
|
||||
|
||||
|
||||
def parse_titles(page_ID: int, page: fitz.Page, json_from_DocXchain_obj: dict, exclude_bboxes):
|
||||
"""
|
||||
:param page_ID: int类型,当前page在当前pdf文档中是第page_D页。
|
||||
:param page :fitz读取的当前页的内容
|
||||
:param res_dir_path: str类型,是每一个pdf文档,在当前.py文件的目录下生成一个与pdf文档同名的文件夹,res_dir_path就是文件夹的dir
|
||||
:param json_from_DocXchain_obj: dict类型,把pdf文档送入DocXChain模型中后,提取bbox,结果保存到pdf文档同名文件夹下的 page_ID.json文件中了。json_from_DocXchain_obj就是打开后的dict
|
||||
"""
|
||||
DPI = 72 # use this resolution
|
||||
pix = page.get_pixmap(dpi=DPI)
|
||||
pageL = 0
|
||||
pageR = int(pix.w)
|
||||
pageU = 0
|
||||
pageD = int(pix.h)
|
||||
|
||||
|
||||
#--------- 通过json_from_DocXchain来获取 title ---------#
|
||||
title_bbox_from_DocXChain = []
|
||||
|
||||
xf_json = json_from_DocXchain_obj
|
||||
width_from_json = xf_json['page_info']['width']
|
||||
height_from_json = xf_json['page_info']['height']
|
||||
LR_scaleRatio = width_from_json / (pageR - pageL)
|
||||
UD_scaleRatio = height_from_json / (pageD - pageU)
|
||||
|
||||
# {0: 'title', # 标题
|
||||
# 1: 'figure', # 图片
|
||||
# 2: 'plain text', # 文本
|
||||
# 3: 'header', # 页眉
|
||||
# 4: 'page number', # 页码
|
||||
# 5: 'footnote', # 脚注
|
||||
# 6: 'footer', # 页脚
|
||||
# 7: 'table', # 表格
|
||||
# 8: 'table caption', # 表格描述
|
||||
# 9: 'figure caption', # 图片描述
|
||||
# 10: 'equation', # 公式
|
||||
# 11: 'full column', # 单栏
|
||||
# 12: 'sub column', # 多栏
|
||||
# 13: 'embedding', # 嵌入公式
|
||||
# 14: 'isolated'} # 单行公式
|
||||
for xf in xf_json['layout_dets']:
|
||||
L = xf['poly'][0] / LR_scaleRatio
|
||||
U = xf['poly'][1] / UD_scaleRatio
|
||||
R = xf['poly'][2] / LR_scaleRatio
|
||||
D = xf['poly'][5] / UD_scaleRatio
|
||||
# L += pageL # 有的页面,artBox偏移了。不在(0,0)
|
||||
# R += pageL
|
||||
# U += pageU
|
||||
# D += pageU
|
||||
L, R = min(L, R), max(L, R)
|
||||
U, D = min(U, D), max(U, D)
|
||||
if xf['category_id'] == 0 and xf['score'] >= 0.3:
|
||||
title_bbox_from_DocXChain.append((L, U, R, D))
|
||||
|
||||
|
||||
title_final_names = []
|
||||
title_final_bboxs = []
|
||||
title_ID = 0
|
||||
for L, U, R, D in title_bbox_from_DocXChain:
|
||||
# cur_title = page.get_pixmap(clip=(L,U,R,D))
|
||||
new_title_name = "title_{}_{}.png".format(page_ID, title_ID) # 标题name
|
||||
# cur_title.save(res_dir_path + '/' + new_title_name) # 把标题存储在新建的文件夹,并命名
|
||||
title_final_names.append(new_title_name) # 把标题的名字存在list中
|
||||
title_final_bboxs.append((L, U, R, D))
|
||||
title_ID += 1
|
||||
|
||||
|
||||
title_final_bboxs.sort(key = lambda LURD: (LURD[1], LURD[0]))
|
||||
curPage_all_title_bboxs = title_final_bboxs
|
||||
return curPage_all_title_bboxs
|
||||
|
||||
@@ -0,0 +1,509 @@
|
||||
import time
|
||||
|
||||
# from anyio import Path
|
||||
|
||||
from libs.commons import fitz, get_delta_time, get_img_s3_client
|
||||
import json
|
||||
import os
|
||||
import math
|
||||
from loguru import logger
|
||||
from layout.bbox_sort import (
|
||||
prepare_bboxes_for_layout_split,
|
||||
)
|
||||
from layout.layout_sort import LAYOUT_UNPROC, get_bboxes_layout, get_columns_cnt_of_layout, sort_text_block
|
||||
from libs.drop_reason import DropReason
|
||||
from libs.markdown_utils import escape_special_markdown_char
|
||||
from libs.safe_filename import sanitize_filename
|
||||
from libs.vis_utils import draw_bbox_on_page, draw_layout_bbox_on_page
|
||||
from pdf2text_recogFigure import parse_images
|
||||
from pdf2text_recogFootnoteLine import remove_headder_footer_one_page # 获取figures的bbox
|
||||
from pdf2text_recogTable import parse_tables # 获取tables的bbox
|
||||
from pdf2text_recogEquation import parse_equations # 获取equations的bbox
|
||||
from pdf2text_recogHeader import parse_headers # 获取headers的bbox
|
||||
from pdf2text_recogPageNo import parse_pageNos # 获取pageNos的bbox
|
||||
from pdf2text_recogFootnote import parse_footnotes_by_model, parse_footnotes_by_rule # 获取footnotes的bbox
|
||||
from pdf2text_recogFooter import parse_footers # 获取footers的bbox
|
||||
|
||||
from pdf2text_recogPara import (
|
||||
ParaProcessPipeline,
|
||||
TitleDetectionException,
|
||||
TitleLevelException,
|
||||
ParaSplitException,
|
||||
ParaMergeException,
|
||||
DenseSingleLineBlockException,
|
||||
)
|
||||
from pre_proc.main_text_font import get_main_text_font
|
||||
from pre_proc.remove_colored_strip_bbox import remove_colored_strip_textblock
|
||||
|
||||
'''
|
||||
from para.para_pipeline import ParaProcessPipeline
|
||||
from para.exceptions import (
|
||||
TitleDetectionException,
|
||||
TitleLevelException,
|
||||
ParaSplitException,
|
||||
ParaMergeException,
|
||||
DenseSingleLineBlockException,
|
||||
)
|
||||
'''
|
||||
|
||||
from libs.commons import read_file, join_path
|
||||
from libs.pdf_image_tools import save_images_by_bboxes
|
||||
from post_proc.footnote_remove import merge_footnote_blocks, remove_footnote_blocks
|
||||
from pre_proc.citationmarker_remove import remove_citation_marker
|
||||
from pre_proc.equations_replace import combine_chars_to_pymudict, remove_chars_in_text_blocks, replace_equations_in_textblock
|
||||
from pre_proc.pdf_filter import pdf_filter
|
||||
from pre_proc.detect_footer_header import drop_footer_header
|
||||
from pre_proc.construct_paras import construct_page_component
|
||||
from pre_proc.image_fix import combine_images, fix_image_vertical, fix_seperated_image, include_img_title
|
||||
from post_proc.pdf_post_filter import pdf_post_filter
|
||||
from pre_proc.remove_rotate_bbox import get_side_boundry, remove_rotate_side_textblock, remove_side_blank_block
|
||||
from pre_proc.resolve_bbox_conflict import check_text_block_horizontal_overlap, resolve_bbox_overlap_conflict
|
||||
from pre_proc.table_fix import fix_table_text_block, fix_tables, include_table_title
|
||||
|
||||
denseSingleLineBlockException_msg = DenseSingleLineBlockException().message
|
||||
titleDetectionException_msg = TitleDetectionException().message
|
||||
titleLevelException_msg = TitleLevelException().message
|
||||
paraSplitException_msg = ParaSplitException().message
|
||||
paraMergeException_msg = ParaMergeException().message
|
||||
|
||||
|
||||
def get_docx_model_output(pdf_model_output, pdf_model_s3_profile, page_id):
|
||||
if isinstance(pdf_model_output, str):
|
||||
model_output_json_path = join_path(pdf_model_output, f"page_{page_id + 1}.json") # 模型输出的页面编号从1开始的
|
||||
if os.path.exists(model_output_json_path):
|
||||
json_from_docx = read_file(model_output_json_path, pdf_model_s3_profile)
|
||||
model_output_json = json.loads(json_from_docx)
|
||||
else:
|
||||
try:
|
||||
model_output_json_path = join_path(pdf_model_output, "model.json")
|
||||
with open(model_output_json_path, "r", encoding="utf-8") as f:
|
||||
model_output_json = json.load(f)
|
||||
model_output_json = model_output_json["doc_layout_result"][page_id]
|
||||
except:
|
||||
s3_model_output_json_path = join_path(pdf_model_output, f"page_{page_id + 1}.json")
|
||||
s3_model_output_json_path = join_path(pdf_model_output, f"{page_id}.json")
|
||||
#s3_model_output_json_path = join_path(pdf_model_output, f"page_{page_id }.json")
|
||||
# logger.warning(f"model_output_json_path: {model_output_json_path} not found. try to load from s3: {s3_model_output_json_path}")
|
||||
|
||||
s = read_file(s3_model_output_json_path, pdf_model_s3_profile)
|
||||
return json.loads(s)
|
||||
|
||||
elif isinstance(pdf_model_output, list):
|
||||
model_output_json = pdf_model_output[page_id]
|
||||
|
||||
return model_output_json
|
||||
|
||||
|
||||
def parse_pdf_by_model(
|
||||
s3_pdf_path,
|
||||
s3_pdf_profile,
|
||||
pdf_model_output,
|
||||
save_path,
|
||||
book_name,
|
||||
pdf_model_profile=None,
|
||||
image_s3_config=None,
|
||||
start_page_id=0,
|
||||
end_page_id=None,
|
||||
junk_img_bojids=[],
|
||||
debug_mode=False,
|
||||
):
|
||||
pdf_bytes = read_file(s3_pdf_path, s3_pdf_profile)
|
||||
save_tmp_path = os.path.join(os.path.dirname(__file__), "..", "..", "tmp", "unittest")
|
||||
md_bookname_save_path = ""
|
||||
book_name = sanitize_filename(book_name)
|
||||
if debug_mode:
|
||||
save_path = join_path(save_tmp_path, "md")
|
||||
pdf_local_path = join_path(save_tmp_path, "download-pdfs", book_name)
|
||||
|
||||
if not os.path.exists(os.path.dirname(pdf_local_path)):
|
||||
# 如果目录不存在,创建它
|
||||
os.makedirs(os.path.dirname(pdf_local_path))
|
||||
|
||||
md_bookname_save_path = join_path(save_tmp_path, "md", book_name)
|
||||
if not os.path.exists(md_bookname_save_path):
|
||||
# 如果目录不存在,创建它
|
||||
os.makedirs(md_bookname_save_path)
|
||||
|
||||
with open(pdf_local_path + ".pdf", "wb") as pdf_file:
|
||||
pdf_file.write(pdf_bytes)
|
||||
|
||||
pdf_docs = fitz.open("pdf", pdf_bytes)
|
||||
pdf_info_dict = {}
|
||||
img_s3_client = get_img_s3_client(save_path, image_s3_config) # 更改函数名和参数,避免歧义
|
||||
# img_s3_client = "img_s3_client" #不创建这个对象,直接用字符串占位
|
||||
|
||||
start_time = time.time()
|
||||
|
||||
"""通过统计pdf全篇文字,识别正文字体"""
|
||||
main_text_font = get_main_text_font(pdf_docs)
|
||||
|
||||
|
||||
end_page_id = end_page_id if end_page_id else len(pdf_docs) - 1
|
||||
for page_id in range(start_page_id, end_page_id + 1):
|
||||
page = pdf_docs[page_id]
|
||||
page_width = page.rect.width
|
||||
page_height = page.rect.height
|
||||
|
||||
if debug_mode:
|
||||
time_now = time.time()
|
||||
logger.info(f"page_id: {page_id}, last_page_cost_time: {get_delta_time(start_time)}")
|
||||
start_time = time_now
|
||||
"""
|
||||
# 通过一个规则,过滤掉单页超过1500非junkimg的pdf
|
||||
# 对单页面非重复id的img数量做统计,如果当前页超过1500则直接return need_drop
|
||||
"""
|
||||
page_imgs = page.get_images()
|
||||
img_counts = 0
|
||||
for img in page_imgs:
|
||||
img_bojid = img[0]
|
||||
if img_bojid in junk_img_bojids: # 判断这个图片在不在junklist中
|
||||
continue # 如果在junklist就不用管了,跳过
|
||||
else:
|
||||
recs = page.get_image_rects(img, transform=True)
|
||||
if recs: # 如果这张图在当前页面有展示
|
||||
img_counts += 1
|
||||
if img_counts >= 1500: # 如果去除了junkimg的影响,单页img仍然超过1500的话,就排除当前pdf
|
||||
logger.warning(
|
||||
f"page_id: {page_id}, img_counts: {img_counts}, drop this pdf: {book_name}, drop_reason: {DropReason.HIGH_COMPUTATIONAL_lOAD_BY_IMGS}"
|
||||
)
|
||||
result = {"need_drop": True, "drop_reason": DropReason.HIGH_COMPUTATIONAL_lOAD_BY_IMGS}
|
||||
if not debug_mode:
|
||||
return result
|
||||
|
||||
"""
|
||||
==================================================================================================================================
|
||||
首先获取基本的block数据,对pdf进行分解,获取图片、表格、公式、text的bbox
|
||||
"""
|
||||
# 解析pdf原始文本block
|
||||
text_raw_blocks = page.get_text(
|
||||
"dict",
|
||||
flags=fitz.TEXTFLAGS_TEXT,
|
||||
)["blocks"]
|
||||
model_output_json = get_docx_model_output(pdf_model_output, pdf_model_profile, page_id)
|
||||
|
||||
# 解析图片
|
||||
image_bboxes = parse_images(page_id, page, model_output_json, junk_img_bojids)
|
||||
image_bboxes = fix_image_vertical(image_bboxes, text_raw_blocks) # 修正图片的位置
|
||||
image_bboxes = fix_seperated_image(image_bboxes) # 合并有边重合的图片
|
||||
image_bboxes = include_img_title(text_raw_blocks, image_bboxes) # 向图片上方和下方寻找title,使用规则进行匹配,暂时只支持英文规则
|
||||
"""此时image_bboxes中可能出现这种情况,水平并列的2个图片,下方分别有各自的子标题,2个子标题下方又有大标题(形如Figxxx),会出现2个图片的bbox都包含了这个大标题,这种情况需要把图片合并"""
|
||||
image_bboxes = combine_images(image_bboxes) # 合并图片
|
||||
|
||||
# 解析表格并对table_bboxes进行位置的微调,防止表格周围的文字被截断
|
||||
table_bboxes = parse_tables(page_id, page, model_output_json)
|
||||
table_bboxes = fix_tables(page, table_bboxes, include_table_title=True, scan_line_num=2) # 修正
|
||||
table_bboxes = fix_table_text_block(text_raw_blocks, table_bboxes) # 修正与text block的关系,某些table修正与pymupdf获取到的table内textblock没有完全包含,因此要进行一次修正。
|
||||
#debug_show_bbox(pdf_docs, page_id, table_bboxes, [], [b['bbox'] for b in text_raw_blocks], join_path(save_path, book_name, f"{book_name}_debug.pdf"), 7)
|
||||
table_bboxes = include_table_title(text_raw_blocks, table_bboxes) # 向table上方和下方寻找title,使用规则进行匹配,暂时只支持英文规则
|
||||
|
||||
# 解析公式
|
||||
equations_inline_bboxes, equations_interline_bboxes = parse_equations(page_id, page, model_output_json)
|
||||
|
||||
"""
|
||||
==================================================================================================================================
|
||||
进入预处理-1阶段
|
||||
-------------------
|
||||
# # 解析标题
|
||||
# title_bboxs = parse_titles(page_id, page, model_output_json)
|
||||
# # 评估Layout是否规整、简单
|
||||
# isSimpleLayout_flag, fullColumn_cnt, subColumn_cnt, curPage_loss = evaluate_pdf_layout(page_id, page, model_output_json)
|
||||
接下来开始进行预处理过程
|
||||
"""
|
||||
|
||||
"""去掉每页的页码、页眉、页脚"""
|
||||
page_no_bboxs = parse_pageNos(page_id, page, model_output_json)
|
||||
header_bboxs = parse_headers(page_id, page, model_output_json)
|
||||
footer_bboxs = parse_footers(page_id, page, model_output_json)
|
||||
image_bboxes, table_bboxes, remain_text_blocks, removed_hdr_foot_txt_block, removed_hdr_foot_img_block, removed_hdr_foot_table = remove_headder_footer_one_page(text_raw_blocks, image_bboxes, table_bboxes, header_bboxs, footer_bboxs, page_no_bboxs, page_width, page_height)
|
||||
|
||||
"""去除页面上半部分长条色块内的文本块"""
|
||||
remain_text_blocks, removed_colored_narrow_strip_background_text_block = remove_colored_strip_textblock(remain_text_blocks, page)
|
||||
|
||||
#debug_show_bbox(pdf_docs, page_id, footnote_bboxes_by_model, [b['bbox'] for b in remain_text_blocks], header_bboxs, join_path(save_path, book_name, f"{book_name}_debug.pdf"), 7)
|
||||
|
||||
"""去掉旋转的文字:水印、垂直排列的文字"""
|
||||
remain_text_blocks, removed_non_horz_text_block = remove_rotate_side_textblock(
|
||||
remain_text_blocks, page_width, page_height
|
||||
) # 去掉水印,非水平文字
|
||||
remain_text_blocks, removed_empty_side_block = remove_side_blank_block(remain_text_blocks, page_width, page_height) # 删除页面四周可能会留下的完全空白的textblock,这种block形成原因未知
|
||||
|
||||
"""出现在图片、表格上的文字块去掉,把层叠的图片单独分离出来,不参与layout的计算"""
|
||||
(
|
||||
image_bboxes,
|
||||
table_bboxes,
|
||||
equations_interline_bboxes,
|
||||
equations_inline_bboxes,
|
||||
remain_text_blocks,
|
||||
text_block_on_image_removed,
|
||||
images_overlap_backup,
|
||||
interline_eq_temp_text_block
|
||||
) = resolve_bbox_overlap_conflict(
|
||||
image_bboxes, table_bboxes, equations_interline_bboxes, equations_inline_bboxes, remain_text_blocks
|
||||
)
|
||||
|
||||
# """去掉footnote, 从文字和图片中"""
|
||||
# # 通过模型识别到的footnote
|
||||
# footnote_bboxes_by_model = parse_footnotes_by_model(page_id, page, model_output_json, md_bookname_save_path,
|
||||
# debug_mode=debug_mode)
|
||||
# # 通过规则识别到的footnote
|
||||
# footnote_bboxes_by_rule = parse_footnotes_by_rule(remain_text_blocks, page_height, page_id)
|
||||
"""
|
||||
==================================================================================================================================
|
||||
"""
|
||||
if debug_mode: # debugmode截图到本地
|
||||
save_path = join_path(save_tmp_path, "md")
|
||||
|
||||
# 把图、表、公式都进行截图,保存到存储上,返回图片路径作为内容
|
||||
image_info, image_backup_info, table_info, inline_eq_info, interline_eq_info = save_images_by_bboxes(
|
||||
book_name,
|
||||
page_id,
|
||||
page,
|
||||
save_path,
|
||||
image_bboxes,
|
||||
images_overlap_backup,
|
||||
table_bboxes,
|
||||
equations_inline_bboxes,
|
||||
equations_interline_bboxes,
|
||||
# 传入img_s3_client
|
||||
img_s3_client,
|
||||
) # 只要表格和图片的截图
|
||||
|
||||
""""以下进入到公式替换环节 """
|
||||
char_level_text_blocks = page.get_text("rawdict", flags=fitz.TEXTFLAGS_TEXT)['blocks']
|
||||
remain_text_blocks = combine_chars_to_pymudict(remain_text_blocks, char_level_text_blocks)# 合并chars
|
||||
remain_text_blocks = remove_citation_marker(remain_text_blocks) # 先把角标去掉
|
||||
|
||||
remain_text_blocks = replace_equations_in_textblock(remain_text_blocks, inline_eq_info, interline_eq_info)
|
||||
remain_text_blocks = remove_chars_in_text_blocks(remain_text_blocks) # 减少中间态数据体积
|
||||
#debug_show_bbox(pdf_docs, page_id, [b['bbox'] for b in inline_eq_info], [b['bbox'] for b in interline_eq_info], [], join_path(save_path, book_name, f"{book_name}_debug.pdf"), 3)
|
||||
|
||||
"""去掉footnote, 从文字和图片中(先去角标再去footnote试试)"""
|
||||
# 通过模型识别到的footnote
|
||||
footnote_bboxes_by_model = parse_footnotes_by_model(page_id, page, model_output_json, md_bookname_save_path, debug_mode=debug_mode)
|
||||
# 通过规则识别到的footnote
|
||||
footnote_bboxes_by_rule = parse_footnotes_by_rule(remain_text_blocks, page_height, page_id, main_text_font)
|
||||
|
||||
"""进入pdf过滤器,去掉一些不合理的pdf"""
|
||||
is_good_pdf, err = pdf_filter(page, remain_text_blocks, table_bboxes, image_bboxes)
|
||||
if not is_good_pdf:
|
||||
logger.warning(f"page_id: {page_id}, drop this pdf: {book_name}, reason: {err}")
|
||||
if not debug_mode:
|
||||
return err
|
||||
|
||||
"""
|
||||
==================================================================================================================================
|
||||
进行版面布局切分和过滤
|
||||
"""
|
||||
"""在切分之前,先检查一下bbox是否有左右重叠的情况,如果有,那么就认为这个pdf暂时没有能力处理好,这种左右重叠的情况大概率是由于pdf里的行间公式、表格没有被正确识别出来造成的 """
|
||||
|
||||
is_text_block_horz_overlap = check_text_block_horizontal_overlap(remain_text_blocks, header_bboxs, footer_bboxs)
|
||||
|
||||
if is_text_block_horz_overlap:
|
||||
# debug_show_bbox(pdf_docs, page_id, [b['bbox'] for b in remain_text_blocks], [], [], join_path(save_path, book_name, f"{book_name}_debug.pdf"), 0)
|
||||
logger.warning(f"page_id: {page_id}, drop this pdf: {book_name}, reason: {DropReason.TEXT_BLCOK_HOR_OVERLAP}")
|
||||
result = {"need_drop": True, "drop_reason": DropReason.TEXT_BLCOK_HOR_OVERLAP}
|
||||
if not debug_mode:
|
||||
return result
|
||||
|
||||
"""统一格式化成一个数据结构用于计算layout"""
|
||||
page_y0 = 0 if len(header_bboxs) == 0 else max([b[3] for b in header_bboxs])
|
||||
page_y1 = page_height if len(footer_bboxs) == 0 else min([b[1] for b in footer_bboxs])
|
||||
left_x, right_x = get_side_boundry(removed_non_horz_text_block, page_width, page_height)
|
||||
page_boundry = [math.floor(left_x), page_y0 + 1, math.ceil(right_x), page_y1 - 1]
|
||||
# 返回的是一个数组,每个元素[x0, y0, x1, y1, block_content, idx_x, idx_y], 初始时候idx_x, idx_y都是None. 对于图片、公式来说,block_content是图片的地址, 对于段落来说,block_content是段落的内容
|
||||
|
||||
all_bboxes = prepare_bboxes_for_layout_split(
|
||||
image_info, image_backup_info, table_info, inline_eq_info, interline_eq_info, remain_text_blocks, page_boundry, page)
|
||||
#debug_show_bbox(pdf_docs, page_id, [], [], all_bboxes, join_path(save_path, book_name, f"{book_name}_debug.pdf"), 1)
|
||||
"""page_y0, page_y1能够过滤掉页眉和页脚,不会算作layout内"""
|
||||
layout_bboxes, layout_tree = get_bboxes_layout(all_bboxes, page_boundry, page_id)
|
||||
|
||||
if len(remain_text_blocks)>0 and len(all_bboxes)>0 and len(layout_bboxes)==0:
|
||||
logger.warning(f"page_id: {page_id}, drop this pdf: {book_name}, reason: {DropReason.CAN_NOT_DETECT_PAGE_LAYOUT}")
|
||||
result = {"need_drop": True, "drop_reason": DropReason.CAN_NOT_DETECT_PAGE_LAYOUT}
|
||||
if not debug_mode:
|
||||
return result
|
||||
|
||||
"""以下去掉复杂的布局和超过2列的布局"""
|
||||
if any([lay["layout_label"] == LAYOUT_UNPROC for lay in layout_bboxes]): # 复杂的布局
|
||||
logger.warning(f"page_id: {page_id}, drop this pdf: {book_name}, reason: {DropReason.COMPLICATED_LAYOUT}")
|
||||
result = {"need_drop": True, "drop_reason": DropReason.COMPLICATED_LAYOUT}
|
||||
if not debug_mode:
|
||||
return result
|
||||
|
||||
layout_column_width = get_columns_cnt_of_layout(layout_tree)
|
||||
if layout_column_width > 2: # 去掉超过2列的布局pdf
|
||||
logger.warning(f"page_id: {page_id}, drop this pdf: {book_name}, reason: {DropReason.TOO_MANY_LAYOUT_COLUMNS}")
|
||||
result = {
|
||||
"need_drop": True,
|
||||
"drop_reason": DropReason.TOO_MANY_LAYOUT_COLUMNS,
|
||||
"extra_info": {"column_cnt": layout_column_width},
|
||||
}
|
||||
if not debug_mode:
|
||||
return result
|
||||
|
||||
|
||||
"""
|
||||
==================================================================================================================================
|
||||
构造出下游需要的数据结构
|
||||
"""
|
||||
remain_text_blocks = remain_text_blocks + interline_eq_temp_text_block # 把计算layout时候临时删除的行间公式再放回去,防止行间公式替换的时候丢失。
|
||||
removed_text_blocks = []
|
||||
removed_text_blocks.extend(removed_hdr_foot_txt_block)
|
||||
# removed_text_blocks.extend(removed_footnote_text_block)
|
||||
removed_text_blocks.extend(text_block_on_image_removed)
|
||||
removed_text_blocks.extend(removed_non_horz_text_block)
|
||||
removed_text_blocks.extend(removed_colored_narrow_strip_background_text_block)
|
||||
|
||||
removed_images = []
|
||||
# removed_images.extend(footnote_imgs)
|
||||
removed_images.extend(removed_hdr_foot_img_block)
|
||||
|
||||
images_backup = []
|
||||
images_backup.extend(image_backup_info)
|
||||
remain_text_blocks = escape_special_markdown_char(remain_text_blocks) # 转义span里的text
|
||||
sorted_text_remain_text_block = sort_text_block(remain_text_blocks, layout_bboxes)
|
||||
|
||||
footnote_bboxes_tmp = []
|
||||
footnote_bboxes_tmp.extend(footnote_bboxes_by_model)
|
||||
footnote_bboxes_tmp.extend(footnote_bboxes_by_rule)
|
||||
|
||||
|
||||
page_info = construct_page_component(
|
||||
page_id,
|
||||
image_info,
|
||||
table_info,
|
||||
sorted_text_remain_text_block,
|
||||
layout_bboxes,
|
||||
inline_eq_info,
|
||||
interline_eq_info,
|
||||
page.get_text("dict", flags=fitz.TEXTFLAGS_TEXT)["blocks"],
|
||||
removed_text_blocks=removed_text_blocks,
|
||||
removed_image_blocks=removed_images,
|
||||
images_backup=images_backup,
|
||||
droped_table_block=[],
|
||||
table_backup=[],
|
||||
layout_tree=layout_tree,
|
||||
page_w=page.rect.width,
|
||||
page_h=page.rect.height,
|
||||
footnote_bboxes_tmp=footnote_bboxes_tmp
|
||||
)
|
||||
pdf_info_dict[f"page_{page_id}"] = page_info
|
||||
|
||||
# end page for
|
||||
|
||||
'''计算后处理阶段耗时'''
|
||||
start_time = time.time()
|
||||
|
||||
"""
|
||||
==================================================================================================================================
|
||||
去掉页眉和页脚,这里需要用到一定的统计量,所以放到最后
|
||||
页眉和页脚主要从文本box和图片box中去除,位于页面的四周。
|
||||
下面函数会直接修改pdf_info_dict,从文字块中、图片中删除属于页眉页脚的内容,删除内容做相对应记录
|
||||
"""
|
||||
# 去页眉页脚
|
||||
header, footer = drop_footer_header(pdf_info_dict)
|
||||
|
||||
"""对单个layout内footnote和他下面的所有textbbox合并"""
|
||||
|
||||
for page_key, page_info in pdf_info_dict.items():
|
||||
page_info = merge_footnote_blocks(page_info, main_text_font)
|
||||
page_info = remove_footnote_blocks(page_info)
|
||||
pdf_info_dict[page_key] = page_info
|
||||
|
||||
"""进入pdf后置过滤器,去掉一些不合理的pdf"""
|
||||
|
||||
i = 0
|
||||
for page_info in pdf_info_dict.values():
|
||||
is_good_pdf, err = pdf_post_filter(page_info)
|
||||
if not is_good_pdf:
|
||||
logger.warning(f"page_id: {i}, drop this pdf: {book_name}, reason: {err}")
|
||||
if not debug_mode:
|
||||
return err
|
||||
i += 1
|
||||
|
||||
if debug_mode:
|
||||
params_file_save_path = join_path(save_tmp_path, "md", book_name, "preproc_out.json")
|
||||
page_draw_rect_save_path = join_path(save_tmp_path, "md", book_name, "layout.pdf")
|
||||
# dir_path = os.path.dirname(page_draw_rect_save_path)
|
||||
# if not os.path.exists(dir_path):
|
||||
# # 如果目录不存在,创建它
|
||||
# os.makedirs(dir_path)
|
||||
|
||||
with open(params_file_save_path, "w", encoding="utf-8") as f:
|
||||
json.dump(pdf_info_dict, f, ensure_ascii=False, indent=4)
|
||||
# 先检测本地 page_draw_rect_save_path 是否存在,如果存在则删除
|
||||
if os.path.exists(page_draw_rect_save_path):
|
||||
os.remove(page_draw_rect_save_path)
|
||||
# 绘制bbox和layout到pdf
|
||||
draw_bbox_on_page(pdf_docs, pdf_info_dict, page_draw_rect_save_path)
|
||||
draw_layout_bbox_on_page(pdf_docs, pdf_info_dict, header, footer, page_draw_rect_save_path)
|
||||
|
||||
if debug_mode:
|
||||
# 打印后处理阶段耗时
|
||||
logger.info(f"post_processing_time: {get_delta_time(start_time)}")
|
||||
|
||||
"""
|
||||
==================================================================================================================================
|
||||
进入段落处理-2阶段
|
||||
"""
|
||||
start_time = time.time()
|
||||
|
||||
para_process_pipeline = ParaProcessPipeline()
|
||||
|
||||
def _deal_with_text_exception(error_info):
|
||||
logger.warning(f"page_id: {page_id}, drop this pdf: {book_name}, reason: {error_info}")
|
||||
if error_info == denseSingleLineBlockException_msg:
|
||||
logger.warning(f"Drop this pdf: {book_name}, reason: {DropReason.DENSE_SINGLE_LINE_BLOCK}")
|
||||
result = {"need_drop": True, "drop_reason": DropReason.DENSE_SINGLE_LINE_BLOCK}
|
||||
return result
|
||||
if error_info == titleDetectionException_msg:
|
||||
logger.warning(f"Drop this pdf: {book_name}, reason: {DropReason.TITLE_DETECTION_FAILED}")
|
||||
result = {"need_drop": True, "drop_reason": DropReason.TITLE_DETECTION_FAILED}
|
||||
return result
|
||||
elif error_info == titleLevelException_msg:
|
||||
logger.warning(f"Drop this pdf: {book_name}, reason: {DropReason.TITLE_LEVEL_FAILED}")
|
||||
result = {"need_drop": True, "drop_reason": DropReason.TITLE_LEVEL_FAILED}
|
||||
return result
|
||||
elif error_info == paraSplitException_msg:
|
||||
logger.warning(f"Drop this pdf: {book_name}, reason: {DropReason.PARA_SPLIT_FAILED}")
|
||||
result = {"need_drop": True, "drop_reason": DropReason.PARA_SPLIT_FAILED}
|
||||
return result
|
||||
elif error_info == paraMergeException_msg:
|
||||
logger.warning(f"Drop this pdf: {book_name}, reason: {DropReason.PARA_MERGE_FAILED}")
|
||||
result = {"need_drop": True, "drop_reason": DropReason.PARA_MERGE_FAILED}
|
||||
return result
|
||||
|
||||
if debug_mode:
|
||||
input_pdf_file = f"{pdf_local_path}.pdf"
|
||||
output_dir = f"{save_path}/{book_name}"
|
||||
output_pdf_file = f"{output_dir}/pdf_annos.pdf"
|
||||
|
||||
"""
|
||||
Call the para_process_pipeline function to process the pdf_info_dict.
|
||||
|
||||
Parameters:
|
||||
para_debug_mode: str or None
|
||||
If para_debug_mode is None, the para_process_pipeline will not keep any intermediate results.
|
||||
If para_debug_mode is "simple", the para_process_pipeline will only keep the annos on the pdf and the final results as a json file.
|
||||
If para_debug_mode is "full", the para_process_pipeline will keep all the intermediate results generated during each step.
|
||||
"""
|
||||
pdf_info_dict, error_info = para_process_pipeline.para_process_pipeline(
|
||||
pdf_info_dict,
|
||||
para_debug_mode="simple",
|
||||
input_pdf_path=input_pdf_file,
|
||||
output_pdf_path=output_pdf_file,
|
||||
)
|
||||
# 打印段落处理阶段耗时
|
||||
logger.info(f"para_process_time: {get_delta_time(start_time)}")
|
||||
|
||||
# debug的时候不return drop信息
|
||||
if error_info is not None:
|
||||
_deal_with_text_exception(error_info)
|
||||
return pdf_info_dict
|
||||
else:
|
||||
pdf_info_dict, error_info = para_process_pipeline.para_process_pipeline(pdf_info_dict)
|
||||
if error_info is not None:
|
||||
return _deal_with_text_exception(error_info)
|
||||
|
||||
return pdf_info_dict
|
||||
@@ -0,0 +1,115 @@
|
||||
from libs.boxbase import _is_in
|
||||
from pdf2text_recogFootnoteLine import remove_footnote_text, remove_footnote_image
|
||||
import collections # 统计库
|
||||
|
||||
|
||||
|
||||
def is_below(bbox1, bbox2):
|
||||
# 如果block1的上边y坐标大于block2的下边y坐标,那么block1在block2下面
|
||||
return bbox1[1] > bbox2[3]
|
||||
|
||||
|
||||
def merge_bboxes(bboxes):
|
||||
# 找出所有blocks的最小x0,最大y1,最大x1,最小y0,这就是合并后的bbox
|
||||
x0 = min(bbox[0] for bbox in bboxes)
|
||||
y0 = min(bbox[1] for bbox in bboxes)
|
||||
x1 = max(bbox[2] for bbox in bboxes)
|
||||
y1 = max(bbox[3] for bbox in bboxes)
|
||||
return [x0, y0, x1, y1]
|
||||
|
||||
|
||||
def merge_footnote_blocks(page_info, main_text_font):
|
||||
page_info['merged_bboxes'] = []
|
||||
for layout in page_info['layout_bboxes']:
|
||||
# 找出layout中的所有footnote blocks和preproc_blocks
|
||||
footnote_bboxes = [block for block in page_info['footnote_bboxes_tmp'] if _is_in(block, layout['layout_bbox'])]
|
||||
# 如果没有footnote_blocks,就跳过这个layout
|
||||
if not footnote_bboxes:
|
||||
continue
|
||||
|
||||
preproc_blocks = [block for block in page_info['preproc_blocks'] if _is_in(block['bbox'], layout['layout_bbox'])]
|
||||
# preproc_bboxes = [block['bbox'] for block in preproc_blocks]
|
||||
font_names = collections.Counter()
|
||||
if len(preproc_blocks) > 0:
|
||||
# 存储每一行的文本块大小的列表
|
||||
line_sizes = []
|
||||
# 存储每个文本块的平均行大小
|
||||
block_sizes = []
|
||||
for block in preproc_blocks:
|
||||
block_line_sizes = []
|
||||
block_fonts = collections.Counter()
|
||||
for line in block['lines']:
|
||||
# 提取每个span的size属性,并计算行大小
|
||||
span_sizes = [span['size'] for span in line['spans'] if 'size' in span]
|
||||
if span_sizes:
|
||||
line_size = sum(span_sizes) / len(span_sizes)
|
||||
line_sizes.append(line_size)
|
||||
block_line_sizes.append(line_size)
|
||||
span_font = [(span['font'], len(span['text'])) for span in line['spans'] if
|
||||
'font' in span and len(span['text']) > 0]
|
||||
if span_font:
|
||||
# # todo main_text_font应该用基于字数最多的字体而不是span级别的统计
|
||||
# font_names.append(font_name for font_name in span_font)
|
||||
# block_fonts.append(font_name for font_name in span_font)
|
||||
for font, count in span_font:
|
||||
# font_names.extend([font] * count)
|
||||
# block_fonts.extend([font] * count)
|
||||
font_names[font] += count
|
||||
block_fonts[font] += count
|
||||
if block_line_sizes:
|
||||
# 计算文本块的平均行大小
|
||||
block_size = sum(block_line_sizes) / len(block_line_sizes)
|
||||
block_font = block_fonts.most_common(1)[0][0]
|
||||
block_sizes.append((block, block_size, block_font))
|
||||
|
||||
# 计算main_text_size
|
||||
# main_text_font = font_names.most_common(1)[0][0]
|
||||
main_text_size = collections.Counter(line_sizes).most_common(1)[0][0]
|
||||
else:
|
||||
continue
|
||||
|
||||
need_merge_bboxes = []
|
||||
# 任何一个下面有正文block的footnote bbox都是假footnote
|
||||
for footnote_bbox in footnote_bboxes:
|
||||
# 检测footnote下面是否有正文block(正文block需满足,block平均size大于等于main_text_size,且block行数大于等于5)
|
||||
main_text_bboxes_below = [block['bbox'] for block, size, block_font in block_sizes if
|
||||
is_below(block['bbox'], footnote_bbox) and
|
||||
sum([size >= main_text_size,
|
||||
len(block['lines']) >= 5,
|
||||
block_font == main_text_font]) >= 2]
|
||||
# 如果main_text_bboxes_below不为空,说明footnote下面有正文block,这个footnote不成立,跳过
|
||||
if len(main_text_bboxes_below) > 0:
|
||||
continue
|
||||
else:
|
||||
# 否则,说明footnote下面没有正文block,这个footnote成立,添加到待merge的footnote_bboxes中
|
||||
need_merge_bboxes.append(footnote_bbox)
|
||||
if len(need_merge_bboxes) == 0:
|
||||
continue
|
||||
# 找出最靠上的footnote block
|
||||
top_footnote_bbox = min(need_merge_bboxes, key=lambda bbox: bbox[1])
|
||||
# 找出所有在top_footnote_block下面的preproc_blocks,并确保这些preproc_blocks的平均行大小小于main_text_size
|
||||
bboxes_below = [block['bbox'] for block, size, block_font in block_sizes if is_below(block['bbox'], top_footnote_bbox)]
|
||||
# # 找出所有在top_footnote_block下面的preproc_blocks
|
||||
# bboxes_below = [bbox for bbox in preproc_bboxes if is_below(bbox, top_footnote_bbox)]
|
||||
# 合并top_footnote_block和blocks_below
|
||||
merged_bbox = merge_bboxes([top_footnote_bbox] + bboxes_below)
|
||||
# 添加到新的footnote_bboxes_tmp中
|
||||
page_info['merged_bboxes'].append(merged_bbox)
|
||||
return page_info
|
||||
|
||||
|
||||
def remove_footnote_blocks(page_info):
|
||||
if page_info.get('merged_bboxes'):
|
||||
# 从文字中去掉footnote
|
||||
remain_text_blocks, removed_footnote_text_blocks = remove_footnote_text(page_info['preproc_blocks'], page_info['merged_bboxes'])
|
||||
# 从图片中去掉footnote
|
||||
image_blocks, removed_footnote_imgs_blocks = remove_footnote_image(page_info['images'], page_info['merged_bboxes'])
|
||||
# 更新page_info
|
||||
page_info['preproc_blocks'] = remain_text_blocks
|
||||
page_info['images'] = image_blocks
|
||||
page_info['droped_text_block'].extend(removed_footnote_text_blocks)
|
||||
page_info['droped_image_block'].extend(removed_footnote_imgs_blocks)
|
||||
# 删除footnote_bboxes_tmp和merged_bboxes
|
||||
del page_info['merged_bboxes']
|
||||
del page_info['footnote_bboxes_tmp']
|
||||
return page_info
|
||||
@@ -0,0 +1,67 @@
|
||||
from loguru import logger
|
||||
|
||||
from layout.layout_sort import get_columns_cnt_of_layout
|
||||
from libs.drop_reason import DropReason
|
||||
|
||||
|
||||
def __is_pseudo_single_column(page_info) -> bool:
|
||||
"""
|
||||
判断一个页面是否伪单列。
|
||||
|
||||
Args:
|
||||
page_info (dict): 页面信息字典,包括'_layout_tree'和'preproc_blocks'。
|
||||
|
||||
Returns:
|
||||
Tuple[bool, Optional[str]]: 如果页面伪单列返回(True, extra_info),否则返回(False, None)。
|
||||
|
||||
"""
|
||||
layout_tree = page_info['_layout_tree']
|
||||
layout_column_width = get_columns_cnt_of_layout(layout_tree)
|
||||
if layout_column_width == 1:
|
||||
text_blocks = page_info['preproc_blocks']
|
||||
# 遍历每一个text_block
|
||||
for text_block in text_blocks:
|
||||
lines = text_block['lines']
|
||||
num_lines = len(lines)
|
||||
num_satisfying_lines = 0
|
||||
|
||||
for i in range(num_lines - 1):
|
||||
current_line = lines[i]
|
||||
next_line = lines[i + 1]
|
||||
|
||||
# 获取当前line和下一个line的bbox属性
|
||||
current_bbox = current_line['bbox']
|
||||
next_bbox = next_line['bbox']
|
||||
|
||||
# 检查是否满足条件
|
||||
if next_bbox[0] > current_bbox[2] or next_bbox[2] < current_bbox[0]:
|
||||
num_satisfying_lines += 1
|
||||
# 如果有一半以上的line满足条件,就drop
|
||||
# print("num_satisfying_lines:", num_satisfying_lines, "num_lines:", num_lines)
|
||||
if num_lines > 20:
|
||||
radio = num_satisfying_lines / num_lines
|
||||
if radio >= 0.5:
|
||||
extra_info = f"{{num_lines: {num_lines}, num_satisfying_lines: {num_satisfying_lines}}}"
|
||||
block_text = []
|
||||
for line in lines:
|
||||
if line['spans']:
|
||||
for span in line['spans']:
|
||||
block_text.append(span['text'])
|
||||
logger.warning(f"pseudo_single_column block_text: {block_text}")
|
||||
return True, extra_info
|
||||
|
||||
return False, None
|
||||
|
||||
|
||||
def pdf_post_filter(page_info) -> tuple:
|
||||
"""
|
||||
return:(True|False, err_msg)
|
||||
True, 如果pdf符合要求
|
||||
False, 如果pdf不符合要求
|
||||
|
||||
"""
|
||||
bool_is_pseudo_single_column, extra_info = __is_pseudo_single_column(page_info)
|
||||
if bool_is_pseudo_single_column:
|
||||
return False, {"need_drop": True, "drop_reason": DropReason.PSEUDO_SINGLE_COLUMN, "extra_info": extra_info}
|
||||
|
||||
return True, None
|
||||
@@ -0,0 +1,148 @@
|
||||
"""
|
||||
去掉正文的引文引用marker
|
||||
https://aicarrier.feishu.cn/wiki/YLOPwo1PGiwFRdkwmyhcZmr0n3d
|
||||
"""
|
||||
import re
|
||||
from loguru import logger
|
||||
from libs.nlp_utils import NLPModels
|
||||
|
||||
|
||||
__NLP_MODEL = NLPModels()
|
||||
|
||||
def check_1(spans, cur_span_i):
|
||||
"""寻找前一个char,如果是句号,逗号,那么就是角标"""
|
||||
if cur_span_i==0:
|
||||
return False # 不是角标
|
||||
pre_span = spans[cur_span_i-1]
|
||||
pre_char = pre_span['chars'][-1]['c']
|
||||
if pre_char in ['。', ',', '.', ',']:
|
||||
return True
|
||||
|
||||
return False
|
||||
|
||||
|
||||
def check_2(spans, cur_span_i):
|
||||
"""检查前面一个span的最后一个单词,如果长度大于5,全都是字母,并且不含大写,就是角标"""
|
||||
pattern = r'\b[A-Z]\.\s[A-Z][a-z]*\b' # 形如A. Bcde, L. Bcde, 人名的缩写
|
||||
|
||||
if cur_span_i==0 and len(spans)>1:
|
||||
next_span = spans[cur_span_i+1]
|
||||
next_txt = "".join([c['c'] for c in next_span['chars']])
|
||||
result = __NLP_MODEL.detect_entity_catgr_using_nlp(next_txt)
|
||||
if result in ["PERSON", "GPE", "ORG"]:
|
||||
return True
|
||||
|
||||
if re.findall(pattern, next_txt):
|
||||
return True
|
||||
|
||||
return False # 不是角标
|
||||
elif cur_span_i==0 and len(spans)==1: # 角标占用了整行?谨慎删除
|
||||
return False
|
||||
|
||||
# 如果这个span是最后一个span,
|
||||
if cur_span_i==len(spans)-1:
|
||||
pre_span = spans[cur_span_i-1]
|
||||
pre_txt = "".join([c['c'] for c in pre_span['chars']])
|
||||
pre_word = pre_txt.split(' ')[-1]
|
||||
result = __NLP_MODEL.detect_entity_catgr_using_nlp(pre_txt)
|
||||
if result in ["PERSON", "GPE", "ORG"]:
|
||||
return True
|
||||
|
||||
if re.findall(pattern, pre_txt):
|
||||
return True
|
||||
|
||||
return len(pre_word) > 5 and pre_word.isalpha() and pre_word.islower()
|
||||
else: # 既不是第一个span,也不是最后一个span,那么此时检查一下这个角标距离前后哪个单词更近就属于谁的角标
|
||||
pre_span = spans[cur_span_i-1]
|
||||
next_span = spans[cur_span_i+1]
|
||||
cur_span = spans[cur_span_i]
|
||||
# 找到前一个和后一个span里的距离最近的单词
|
||||
pre_distance = 10000 # 一个很大的数
|
||||
next_distance = 10000 # 一个很大的数
|
||||
for c in pre_span['chars'][::-1]:
|
||||
if c['c'].isalpha():
|
||||
pre_distance = cur_span['bbox'][0] - c['bbox'][2]
|
||||
break
|
||||
for c in next_span['chars']:
|
||||
if c['c'].isalpha():
|
||||
next_distance = c['bbox'][0] - cur_span['bbox'][2]
|
||||
break
|
||||
|
||||
if pre_distance<next_distance:
|
||||
belong_to_span = pre_span
|
||||
else:
|
||||
belong_to_span = next_span
|
||||
|
||||
txt = "".join([c['c'] for c in belong_to_span['chars']])
|
||||
pre_word = txt.split(' ')[-1]
|
||||
result = __NLP_MODEL.detect_entity_catgr_using_nlp(txt)
|
||||
if result in ["PERSON", "GPE", "ORG"]:
|
||||
return True
|
||||
|
||||
if re.findall(pattern, txt):
|
||||
return True
|
||||
|
||||
return len(pre_word) > 5 and pre_word.isalpha() and pre_word.islower()
|
||||
|
||||
|
||||
def check_3(spans, cur_span_i):
|
||||
"""上标里有[], 有*, 有-, 有逗号"""
|
||||
# 如[2-3],[22]
|
||||
# 如 2,3,4
|
||||
cur_span_txt = ''.join(c['c'] for c in spans[cur_span_i]['chars']).strip()
|
||||
bad_char = ['[', ']', '*', ',']
|
||||
|
||||
if any([c in cur_span_txt for c in bad_char]) and any(character.isdigit() for character in cur_span_txt):
|
||||
return True
|
||||
|
||||
# 如2-3, a-b
|
||||
patterns = [r'\d+-\d+', r'[a-zA-Z]-[a-zA-Z]', r'[a-zA-Z],[a-zA-Z]']
|
||||
for pattern in patterns:
|
||||
match = re.match(pattern, cur_span_txt)
|
||||
if match is not None:
|
||||
return True
|
||||
|
||||
return False
|
||||
|
||||
|
||||
def remove_citation_marker(with_char_text_blcoks):
|
||||
for blk in with_char_text_blcoks:
|
||||
for line in blk['lines']:
|
||||
# 如果span里的个数少于2个,那只能忽略,角标不可能自己独占一行
|
||||
if len(line['spans'])<=1:
|
||||
continue
|
||||
|
||||
# 找到高度最高的span作为位置比较的基准
|
||||
max_hi_span = line['spans'][0]['bbox']
|
||||
min_font_sz = 10000
|
||||
for s in line['spans']:
|
||||
if max_hi_span[3]-max_hi_span[1]<s['bbox'][3]-s['bbox'][1]:
|
||||
max_hi_span = s['bbox']
|
||||
if min_font_sz>s['size']:
|
||||
min_font_sz = s['size']
|
||||
|
||||
base_span_mid_y = (max_hi_span[3]+max_hi_span[1])/2
|
||||
|
||||
|
||||
span_to_del = []
|
||||
for i, span in enumerate(line['spans']):
|
||||
span_hi = span['bbox'][3]-span['bbox'][1]
|
||||
span_mid_y = (span['bbox'][3]+span['bbox'][1])/2
|
||||
span_font_sz = span['size']
|
||||
|
||||
if (base_span_mid_y-span_mid_y)/span_hi>0.2 or (base_span_mid_y-span_mid_y>0 and abs(span_font_sz-min_font_sz)/min_font_sz<0.1):
|
||||
"""
|
||||
1. 它的前一个char如果是句号或者逗号的话,那么肯定是角标而不是公式
|
||||
2. 如果这个角标的前面是一个单词(长度大于5)而不是任何大写或小写的短字母的话 应该也是角标
|
||||
3. 上标里有数字和逗号或者数字+星号的组合,方括号,一般肯定就是角标了
|
||||
4. 这个角标属于前文还是后文要根据距离来判断,如果距离前面的文本太近,那么就是前面的角标,否则就是后面的角标
|
||||
"""
|
||||
if check_1(line['spans'], i) or check_2(line['spans'], i) or check_3(line['spans'], i):
|
||||
"""删除掉这个角标:删除这个span, 同时还要更新line的text"""
|
||||
span_to_del.append(span)
|
||||
if len(span_to_del)>0:
|
||||
for span in span_to_del:
|
||||
line['spans'].remove(span)
|
||||
line['text'] = ''.join([c['c'] for s in line['spans'] for c in s['chars']])
|
||||
|
||||
return with_char_text_blcoks
|
||||
@@ -0,0 +1,30 @@
|
||||
|
||||
def construct_page_component(page_id, image_info, table_info, text_blocks_preproc, layout_bboxes, inline_eq_info, interline_eq_info, raw_pymu_blocks,
|
||||
removed_text_blocks, removed_image_blocks, images_backup, droped_table_block, table_backup,layout_tree,
|
||||
page_w, page_h, footnote_bboxes_tmp):
|
||||
"""
|
||||
|
||||
"""
|
||||
return_dict = {}
|
||||
|
||||
return_dict['para_blocks'] = {}
|
||||
return_dict['preproc_blocks'] = text_blocks_preproc
|
||||
return_dict['images'] = image_info
|
||||
return_dict['tables'] = table_info
|
||||
return_dict['interline_equations'] = interline_eq_info
|
||||
return_dict['inline_equations'] = inline_eq_info
|
||||
return_dict['layout_bboxes'] = layout_bboxes
|
||||
return_dict['pymu_raw_blocks'] = raw_pymu_blocks
|
||||
return_dict['global_statistic'] = {}
|
||||
|
||||
return_dict['droped_text_block'] = removed_text_blocks
|
||||
return_dict['droped_image_block'] = removed_image_blocks
|
||||
return_dict['droped_table_block'] = []
|
||||
return_dict['image_backup'] = images_backup
|
||||
return_dict['table_backup'] = []
|
||||
return_dict['page_idx'] = page_id
|
||||
return_dict['page_size'] = [page_w, page_h]
|
||||
return_dict['_layout_tree'] = layout_tree # 辅助分析layout作用
|
||||
return_dict['footnote_bboxes_tmp'] = footnote_bboxes_tmp
|
||||
|
||||
return return_dict
|
||||
@@ -0,0 +1,286 @@
|
||||
from collections import defaultdict
|
||||
|
||||
from loguru import logger
|
||||
|
||||
from libs.boxbase import _is_in, calculate_iou
|
||||
|
||||
|
||||
def compare_bbox_with_list(bbox, bbox_list, tolerance=1):
|
||||
return any(all(abs(a - b) < tolerance for a, b in zip(bbox, common_bbox)) for common_bbox in bbox_list)
|
||||
|
||||
def is_single_line_block(block):
|
||||
# Determine based on the width and height of the block
|
||||
block_width = block["X1"] - block["X0"]
|
||||
block_height = block["bbox"][3] - block["bbox"][1]
|
||||
|
||||
# If the height of the block is close to the average character height and the width is large, it is considered a single line
|
||||
return block_height <= block["avg_char_height"] * 3 and block_width > block["avg_char_width"] * 3
|
||||
|
||||
def get_most_common_bboxes(bboxes, page_height, position="top", threshold=0.25, num_bboxes=3, min_frequency=2):
|
||||
"""
|
||||
This function gets the most common bboxes from the bboxes
|
||||
|
||||
Parameters
|
||||
----------
|
||||
bboxes : list
|
||||
bboxes
|
||||
page_height : float
|
||||
height of the page
|
||||
position : str, optional
|
||||
"top" or "bottom", by default "top"
|
||||
threshold : float, optional
|
||||
threshold, by default 0.25
|
||||
num_bboxes : int, optional
|
||||
number of bboxes to return, by default 3
|
||||
min_frequency : int, optional
|
||||
minimum frequency of the bbox, by default 2
|
||||
|
||||
Returns
|
||||
-------
|
||||
common_bboxes : list
|
||||
common bboxes
|
||||
"""
|
||||
# Filter bbox by position
|
||||
if position == "top":
|
||||
filtered_bboxes = [bbox for bbox in bboxes if bbox[1] < page_height * threshold]
|
||||
else:
|
||||
filtered_bboxes = [bbox for bbox in bboxes if bbox[3] > page_height * (1 - threshold)]
|
||||
|
||||
# Find the most common bbox
|
||||
bbox_count = defaultdict(int)
|
||||
for bbox in filtered_bboxes:
|
||||
bbox_count[tuple(bbox)] += 1
|
||||
|
||||
# Get the most frequently occurring bbox, but only consider it when the frequency exceeds min_frequency
|
||||
common_bboxes = [
|
||||
bbox for bbox, count in sorted(bbox_count.items(), key=lambda item: item[1], reverse=True) if count >= min_frequency
|
||||
][:num_bboxes]
|
||||
return common_bboxes
|
||||
|
||||
def detect_footer_header2(result_dict, similarity_threshold=0.5):
|
||||
"""
|
||||
This function detects the header and footer of the document.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
result_dict : dict
|
||||
result dictionary
|
||||
|
||||
Returns
|
||||
-------
|
||||
result_dict : dict
|
||||
result dictionary
|
||||
"""
|
||||
# Traverse all blocks in the document
|
||||
single_line_blocks = 0
|
||||
total_blocks = 0
|
||||
single_line_blocks = 0
|
||||
|
||||
for page_id, blocks in result_dict.items():
|
||||
if page_id.startswith("page_"):
|
||||
for block_key, block in blocks.items():
|
||||
if block_key.startswith("block_"):
|
||||
total_blocks += 1
|
||||
if is_single_line_block(block):
|
||||
single_line_blocks += 1
|
||||
|
||||
# If there are no blocks, skip the header and footer detection
|
||||
if total_blocks == 0:
|
||||
print("No blocks found. Skipping header/footer detection.")
|
||||
return result_dict
|
||||
|
||||
# If most of the blocks are single-line, skip the header and footer detection
|
||||
if single_line_blocks / total_blocks > 0.5: # 50% of the blocks are single-line
|
||||
# print("Skipping header/footer detection for text-dense document.")
|
||||
return result_dict
|
||||
|
||||
# Collect the bounding boxes of all blocks
|
||||
all_bboxes = []
|
||||
all_texts = []
|
||||
|
||||
for page_id, blocks in result_dict.items():
|
||||
if page_id.startswith("page_"):
|
||||
for block_key, block in blocks.items():
|
||||
if block_key.startswith("block_"):
|
||||
all_bboxes.append(block["bbox"])
|
||||
|
||||
# Get the height of the page
|
||||
page_height = max(bbox[3] for bbox in all_bboxes)
|
||||
|
||||
# Get the most common bbox lists for headers and footers
|
||||
common_header_bboxes = get_most_common_bboxes(all_bboxes, page_height, position="top") if all_bboxes else []
|
||||
common_footer_bboxes = get_most_common_bboxes(all_bboxes, page_height, position="bottom") if all_bboxes else []
|
||||
|
||||
# Detect and mark headers and footers
|
||||
for page_id, blocks in result_dict.items():
|
||||
if page_id.startswith("page_"):
|
||||
for block_key, block in blocks.items():
|
||||
if block_key.startswith("block_"):
|
||||
bbox = block["bbox"]
|
||||
text = block["text"]
|
||||
|
||||
is_header = compare_bbox_with_list(bbox, common_header_bboxes)
|
||||
is_footer = compare_bbox_with_list(bbox, common_footer_bboxes)
|
||||
block["is_header"] = int(is_header)
|
||||
block["is_footer"] = int(is_footer)
|
||||
|
||||
return result_dict
|
||||
|
||||
|
||||
def __get_page_size(page_sizes:list):
|
||||
"""
|
||||
页面大小可能不一样
|
||||
"""
|
||||
w = sum([w for w,h in page_sizes])/len(page_sizes)
|
||||
h = sum([h for w,h in page_sizes])/len(page_sizes)
|
||||
return w, h
|
||||
|
||||
def __calculate_iou(bbox1, bbox2):
|
||||
iou = calculate_iou(bbox1, bbox2)
|
||||
return iou
|
||||
|
||||
def __is_same_pos(box1, box2, iou_threshold):
|
||||
iou = __calculate_iou(box1, box2)
|
||||
return iou >= iou_threshold
|
||||
|
||||
|
||||
def get_most_common_bbox(bboxes:list, page_size:list, page_cnt:int, page_range_threshold=0.2, iou_threshold=0.9):
|
||||
"""
|
||||
common bbox必须大于page_cnt的1/3
|
||||
"""
|
||||
min_occurance_cnt = max(3, page_cnt//4)
|
||||
header_det_bbox = []
|
||||
footer_det_bbox = []
|
||||
|
||||
hdr_same_pos_group = []
|
||||
btn_same_pos_group = []
|
||||
|
||||
page_w, page_h = __get_page_size(page_size)
|
||||
top_y, bottom_y = page_w*page_range_threshold, page_h*(1-page_range_threshold)
|
||||
|
||||
top_bbox = [b for b in bboxes if b[3]<top_y]
|
||||
bottom_bbox = [b for b in bboxes if b[1]>bottom_y]
|
||||
# 然后开始排序,寻找最经常出现的bbox, 寻找的时候如果IOU>iou_threshold就算是一个
|
||||
for i in range(0, len(top_bbox)):
|
||||
hdr_same_pos_group.append([top_bbox[i]])
|
||||
for j in range(i+1, len(top_bbox)):
|
||||
if __is_same_pos(top_bbox[i], top_bbox[j], iou_threshold):
|
||||
#header_det_bbox = [min(top_bbox[i][0], top_bbox[j][0]), min(top_bbox[i][1], top_bbox[j][1]), max(top_bbox[i][2], top_bbox[j][2]), max(top_bbox[i][3],top_bbox[j][3])]
|
||||
hdr_same_pos_group[i].append(top_bbox[j])
|
||||
|
||||
for i in range(0, len(bottom_bbox)):
|
||||
btn_same_pos_group.append([bottom_bbox[i]])
|
||||
for j in range(i+1, len(bottom_bbox)):
|
||||
if __is_same_pos(bottom_bbox[i], bottom_bbox[j], iou_threshold):
|
||||
#footer_det_bbox = [min(bottom_bbox[i][0], bottom_bbox[j][0]), min(bottom_bbox[i][1], bottom_bbox[j][1]), max(bottom_bbox[i][2], bottom_bbox[j][2]), max(bottom_bbox[i][3],bottom_bbox[j][3])]
|
||||
btn_same_pos_group[i].append(bottom_bbox[j])
|
||||
|
||||
# 然后看下每一组的bbox,是否符合大于page_cnt一定比例
|
||||
hdr_same_pos_group = [g for g in hdr_same_pos_group if len(g)>=min_occurance_cnt]
|
||||
btn_same_pos_group = [g for g in btn_same_pos_group if len(g)>=min_occurance_cnt]
|
||||
|
||||
# 平铺2个list[list]
|
||||
hdr_same_pos_group = [bbox for g in hdr_same_pos_group for bbox in g]
|
||||
btn_same_pos_group = [bbox for g in btn_same_pos_group for bbox in g]
|
||||
# 寻找hdr_same_pos_group中的box[3]最大值,btn_same_pos_group中的box[1]最小值
|
||||
hdr_same_pos_group.sort(key=lambda b:b[3])
|
||||
btn_same_pos_group.sort(key=lambda b:b[1])
|
||||
|
||||
hdr_y = hdr_same_pos_group[-1][3] if hdr_same_pos_group else 0
|
||||
btn_y = btn_same_pos_group[0][1] if btn_same_pos_group else page_h
|
||||
|
||||
header_det_bbox = [0, 0, page_w, hdr_y]
|
||||
footer_det_bbox = [0, btn_y, page_w, page_h]
|
||||
# logger.warning(f"header: {header_det_bbox}, footer: {footer_det_bbox}")
|
||||
return header_det_bbox, footer_det_bbox, page_w, page_h
|
||||
|
||||
|
||||
def drop_footer_header(pdf_info_dict:dict):
|
||||
"""
|
||||
启用规则探测,在全局的视角上通过统计的方法。
|
||||
"""
|
||||
header = []
|
||||
footer = []
|
||||
|
||||
all_text_bboxes = [blk['bbox'] for _, val in pdf_info_dict.items() for blk in val['preproc_blocks']]
|
||||
image_bboxes = [img['bbox'] for _, val in pdf_info_dict.items() for img in val['images']] + [img['bbox'] for _, val in pdf_info_dict.items() for img in val['image_backup']]
|
||||
page_size = [val['page_size'] for _, val in pdf_info_dict.items()]
|
||||
page_cnt = len(pdf_info_dict.keys()) # 一共多少页
|
||||
header, footer, page_w, page_h = get_most_common_bbox(all_text_bboxes+image_bboxes, page_size, page_cnt)
|
||||
|
||||
""""
|
||||
把范围扩展到页面水平的整个方向上
|
||||
"""
|
||||
if header:
|
||||
header = [0, 0, page_w, header[3]+1]
|
||||
|
||||
if footer:
|
||||
footer = [0, footer[1]-1, page_w, page_h]
|
||||
|
||||
# 找到footer, header范围之后,针对每一页pdf,从text、图片中删除这些范围内的内容
|
||||
# 移除text block
|
||||
|
||||
for _, page_info in pdf_info_dict.items():
|
||||
header_text_blk = []
|
||||
footer_text_blk = []
|
||||
for blk in page_info['preproc_blocks']:
|
||||
blk_bbox = blk['bbox']
|
||||
if header and blk_bbox[3]<=header[3]:
|
||||
blk['tag'] = "header"
|
||||
header_text_blk.append(blk)
|
||||
elif footer and blk_bbox[1]>=footer[1]:
|
||||
blk['tag'] = "footer"
|
||||
footer_text_blk.append(blk)
|
||||
|
||||
# 放入text_block_droped中
|
||||
page_info['droped_text_block'].extend(header_text_blk)
|
||||
page_info['droped_text_block'].extend(footer_text_blk)
|
||||
|
||||
for blk in header_text_blk:
|
||||
page_info['preproc_blocks'].remove(blk)
|
||||
for blk in footer_text_blk:
|
||||
page_info['preproc_blocks'].remove(blk)
|
||||
|
||||
"""接下来把footer、header上的图片也删除掉。图片包括正常的和backup的"""
|
||||
header_image = []
|
||||
footer_image = []
|
||||
|
||||
for image_info in page_info['images']:
|
||||
img_bbox = image_info['bbox']
|
||||
if header and img_bbox[3]<=header[3]:
|
||||
image_info['tag'] = "header"
|
||||
header_image.append(image_info)
|
||||
elif footer and img_bbox[1]>=footer[1]:
|
||||
image_info['tag'] = "footer"
|
||||
footer_image.append(image_info)
|
||||
|
||||
page_info['droped_image_block'].extend(header_image)
|
||||
page_info['droped_image_block'].extend(footer_image)
|
||||
|
||||
for img in header_image:
|
||||
page_info['images'].remove(img)
|
||||
for img in footer_image:
|
||||
page_info['images'].remove(img)
|
||||
|
||||
"""接下来吧backup的图片也删除掉"""
|
||||
header_image = []
|
||||
footer_image = []
|
||||
|
||||
for image_info in page_info['image_backup']:
|
||||
img_bbox = image_info['bbox']
|
||||
if header and img_bbox[3]<=header[3]:
|
||||
image_info['tag'] = "header"
|
||||
header_image.append(image_info)
|
||||
elif footer and img_bbox[1]>=footer[1]:
|
||||
image_info['tag'] = "footer"
|
||||
footer_image.append(image_info)
|
||||
|
||||
page_info['droped_image_block'].extend(header_image)
|
||||
page_info['droped_image_block'].extend(footer_image)
|
||||
|
||||
for img in header_image:
|
||||
page_info['image_backup'].remove(img)
|
||||
for img in footer_image:
|
||||
page_info['image_backup'].remove(img)
|
||||
|
||||
return header, footer
|
||||
@@ -0,0 +1,482 @@
|
||||
"""
|
||||
对pymupdf返回的结构里的公式进行替换,替换为模型识别的公式结果
|
||||
"""
|
||||
import fitz
|
||||
import json
|
||||
import os
|
||||
from pathlib import Path
|
||||
from loguru import logger
|
||||
|
||||
TYPE_INLINE_EQUATION = "inline-equation"
|
||||
TYPE_INTERLINE_EQUATION = "interline-equation"
|
||||
|
||||
|
||||
def combine_chars_to_pymudict(block_dict, char_dict):
|
||||
"""
|
||||
把block级别的pymupdf 结构里加入char结构
|
||||
"""
|
||||
# 因为block_dict 被裁剪过,因此先把他和char_dict文字块对齐,才能进行补充
|
||||
char_map = {tuple(item['bbox']):item for item in char_dict}
|
||||
|
||||
for i in range(len(block_dict)): # blcok
|
||||
block = block_dict[i]
|
||||
key = block['bbox']
|
||||
char_dict_item = char_map[tuple(key)]
|
||||
char_dict_map = {tuple(item['bbox']):item for item in char_dict_item['lines']}
|
||||
for j in range(len(block['lines'])):
|
||||
lines = block['lines'][j]
|
||||
with_char_lines = char_dict_map[lines['bbox']]
|
||||
for k in range(len(lines['spans'])):
|
||||
spans = lines['spans'][k]
|
||||
try:
|
||||
chars = with_char_lines['spans'][k]['chars']
|
||||
except Exception as e:
|
||||
logger.error(char_dict[i]['lines'][j])
|
||||
|
||||
spans['chars'] = chars
|
||||
|
||||
return block_dict
|
||||
|
||||
|
||||
def calculate_overlap_area_2_minbox_area_ratio(bbox1, min_bbox):
|
||||
"""
|
||||
计算box1和box2的重叠面积占最小面积的box的比例
|
||||
"""
|
||||
# Determine the coordinates of the intersection rectangle
|
||||
x_left = max(bbox1[0], min_bbox[0])
|
||||
y_top = max(bbox1[1], min_bbox[1])
|
||||
x_right = min(bbox1[2], min_bbox[2])
|
||||
y_bottom = min(bbox1[3], min_bbox[3])
|
||||
|
||||
if x_right < x_left or y_bottom < y_top:
|
||||
return 0.0
|
||||
|
||||
# The area of overlap area
|
||||
intersection_area = (x_right - x_left) * (y_bottom - y_top)
|
||||
min_box_area = (min_bbox[3]-min_bbox[1])*(min_bbox[2]-min_bbox[0])
|
||||
if min_box_area==0:
|
||||
return 0
|
||||
else:
|
||||
return intersection_area / min_box_area
|
||||
|
||||
|
||||
def _is_xin(bbox1, bbox2):
|
||||
area1 = abs(bbox1[2]-bbox1[0])*abs(bbox1[3]-bbox1[1])
|
||||
area2 = abs(bbox2[2]-bbox2[0])*abs(bbox2[3]-bbox2[1])
|
||||
if area1<area2:
|
||||
ratio = calculate_overlap_area_2_minbox_area_ratio(bbox2, bbox1)
|
||||
else:
|
||||
ratio = calculate_overlap_area_2_minbox_area_ratio(bbox1, bbox2)
|
||||
|
||||
return ratio>0.6
|
||||
|
||||
|
||||
|
||||
def remove_text_block_in_interline_equation_bbox(interline_bboxes, text_blocks):
|
||||
"""消除掉整个块都在行间公式块内部的文本块"""
|
||||
for eq_bbox in interline_bboxes:
|
||||
removed_txt_blk = []
|
||||
for text_blk in text_blocks:
|
||||
text_bbox = text_blk['bbox']
|
||||
if calculate_overlap_area_2_minbox_area_ratio(eq_bbox['bbox'], text_bbox)>=0.7:
|
||||
removed_txt_blk.append(text_blk)
|
||||
for blk in removed_txt_blk:
|
||||
text_blocks.remove(blk)
|
||||
|
||||
return text_blocks
|
||||
|
||||
|
||||
|
||||
def _is_in_or_part_overlap(box1, box2) -> bool:
|
||||
"""
|
||||
两个bbox是否有部分重叠或者包含
|
||||
"""
|
||||
if box1 is None or box2 is None:
|
||||
return False
|
||||
|
||||
x0_1, y0_1, x1_1, y1_1 = box1
|
||||
x0_2, y0_2, x1_2, y1_2 = box2
|
||||
|
||||
return not (x1_1 < x0_2 or # box1在box2的左边
|
||||
x0_1 > x1_2 or # box1在box2的右边
|
||||
y1_1 < y0_2 or # box1在box2的上边
|
||||
y0_1 > y1_2) # box1在box2的下边
|
||||
|
||||
|
||||
def remove_text_block_overlap_interline_equation_bbox(interline_eq_bboxes, pymu_block_list):
|
||||
"""消除掉行行内公式有部分重叠的文本块的内容。
|
||||
同时重新计算消除重叠之后文本块的大小"""
|
||||
deleted_block = []
|
||||
for text_block in pymu_block_list:
|
||||
deleted_line = []
|
||||
for line in text_block['lines']:
|
||||
deleted_span = []
|
||||
for span in line['spans']:
|
||||
deleted_chars = []
|
||||
for char in span['chars']:
|
||||
if any([_is_in_or_part_overlap(char['bbox'], eq_bbox['bbox']) for eq_bbox in interline_eq_bboxes]):
|
||||
deleted_chars.append(char)
|
||||
# 检查span里没有char则删除这个span
|
||||
for char in deleted_chars:
|
||||
span['chars'].remove(char)
|
||||
# 重新计算这个span的大小
|
||||
if len(span['chars'])==0: # 删除这个span
|
||||
deleted_span.append(span)
|
||||
else:
|
||||
span['bbox'] = min([b['bbox'][0] for b in span['chars']]),min([b['bbox'][1] for b in span['chars']]),max([b['bbox'][2] for b in span['chars']]), max([b['bbox'][3] for b in span['chars']])
|
||||
|
||||
# 检查这个span
|
||||
for span in deleted_span:
|
||||
line['spans'].remove(span)
|
||||
if len(line['spans'])==0: #删除这个line
|
||||
deleted_line.append(line)
|
||||
else:
|
||||
line['bbox'] = min([b['bbox'][0] for b in line['spans']]),min([b['bbox'][1] for b in line['spans']]),max([b['bbox'][2] for b in line['spans']]), max([b['bbox'][3] for b in line['spans']])
|
||||
|
||||
# 检查这个block是否可以删除
|
||||
for line in deleted_line:
|
||||
text_block['lines'].remove(line)
|
||||
if len(text_block['lines'])==0: # 删除block
|
||||
deleted_block.append(text_block)
|
||||
else:
|
||||
text_block['bbox'] = min([b['bbox'][0] for b in text_block['lines']]),min([b['bbox'][1] for b in text_block['lines']]),max([b['bbox'][2] for b in text_block['lines']]), max([b['bbox'][3] for b in text_block['lines']])
|
||||
|
||||
# 检查text block删除
|
||||
for block in deleted_block:
|
||||
pymu_block_list.remove(block)
|
||||
if len(pymu_block_list)==0:
|
||||
return []
|
||||
|
||||
return pymu_block_list
|
||||
|
||||
|
||||
def insert_interline_equations_textblock(interline_eq_bboxes, pymu_block_list):
|
||||
"""在行间公式对应的地方插上一个伪造的block"""
|
||||
for eq in interline_eq_bboxes:
|
||||
bbox = eq['bbox']
|
||||
latex_content = eq['latex_text']
|
||||
text_block = {
|
||||
"number": len(pymu_block_list),
|
||||
"type": 0,
|
||||
"bbox": bbox,
|
||||
"lines": [
|
||||
{
|
||||
"spans": [
|
||||
{
|
||||
"size": 9.962599754333496,
|
||||
"_type": TYPE_INTERLINE_EQUATION,
|
||||
"flags": 4,
|
||||
"font": TYPE_INTERLINE_EQUATION,
|
||||
"color": 0,
|
||||
"ascender": 0.9409999847412109,
|
||||
"descender": -0.3050000071525574,
|
||||
"text": f"\n$$\n{latex_content}\n$$\n",
|
||||
"origin": [
|
||||
bbox[0],
|
||||
bbox[1]
|
||||
],
|
||||
"bbox": bbox
|
||||
}
|
||||
],
|
||||
"wmode": 0,
|
||||
"dir": [
|
||||
1.0,
|
||||
0.0
|
||||
],
|
||||
"bbox": bbox
|
||||
}
|
||||
]
|
||||
}
|
||||
pymu_block_list.append(text_block)
|
||||
|
||||
def x_overlap_ratio(box1, box2):
|
||||
a, _, c, _ = box1
|
||||
e, _, g, _ = box2
|
||||
|
||||
# 计算重叠宽度
|
||||
overlap_x = max(min(c, g) - max(a, e), 0)
|
||||
|
||||
# 计算box1的宽度
|
||||
width1 = g - e
|
||||
|
||||
# 计算重叠比例
|
||||
overlap_ratio = overlap_x / width1 if width1 != 0 else 0
|
||||
|
||||
return overlap_ratio
|
||||
|
||||
def __is_x_dir_overlap(bbox1, bbox2):
|
||||
return not (bbox1[2]<bbox2[0] or bbox1[0]>bbox2[2])
|
||||
|
||||
def __y_overlap_ratio(box1, box2):
|
||||
""""""
|
||||
_, b, _, d = box1
|
||||
_, f, _, h = box2
|
||||
|
||||
# 计算重叠高度
|
||||
overlap_y = max(min(d, h) - max(b, f), 0)
|
||||
|
||||
# 计算box1的高度
|
||||
height1 = d - b
|
||||
|
||||
# 计算重叠比例
|
||||
overlap_ratio = overlap_y / height1 if height1 != 0 else 0
|
||||
|
||||
return overlap_ratio
|
||||
|
||||
def replace_line_v2(eqinfo, line):
|
||||
"""
|
||||
扫描这一行所有的和公式框X方向重叠的char,然后计算char的左、右x0, x1,位于这个区间内的span删除掉。
|
||||
最后与这个x0,x1有相交的span0, span1内部进行分割。
|
||||
"""
|
||||
first_overlap_span = -1
|
||||
first_overlap_span_idx = -1
|
||||
last_overlap_span = -1
|
||||
delete_chars = []
|
||||
for i in range(0, len(line['spans'])):
|
||||
if line['spans'][i].get("_type", None) is not None:
|
||||
continue # 忽略,因为已经是插入的伪造span公式了
|
||||
|
||||
for char in line['spans'][i]['chars']:
|
||||
if __is_x_dir_overlap(eqinfo['bbox'], char['bbox']):
|
||||
line_txt = ""
|
||||
for span in line['spans']:
|
||||
span_txt = "<span>"
|
||||
for ch in span['chars']:
|
||||
span_txt = span_txt + ch['c']
|
||||
|
||||
span_txt = span_txt + "</span>"
|
||||
|
||||
line_txt = line_txt + span_txt
|
||||
|
||||
if first_overlap_span_idx == -1:
|
||||
first_overlap_span = line['spans'][i]
|
||||
first_overlap_span_idx = i
|
||||
last_overlap_span = line['spans'][i]
|
||||
delete_chars.append(char)
|
||||
|
||||
# 第一个和最后一个char要进行检查,到底属于公式多还是属于正常span多
|
||||
if len(delete_chars)>0:
|
||||
ch0_bbox = delete_chars[0]['bbox']
|
||||
if x_overlap_ratio(eqinfo['bbox'], ch0_bbox)<0.51:
|
||||
delete_chars.remove(delete_chars[0])
|
||||
if len(delete_chars)>0:
|
||||
ch0_bbox = delete_chars[-1]['bbox']
|
||||
if x_overlap_ratio(eqinfo['bbox'], ch0_bbox)<0.51:
|
||||
delete_chars.remove(delete_chars[-1])
|
||||
|
||||
# 计算x方向上被删除区间内的char的真实x0, x1
|
||||
if len(delete_chars):
|
||||
x0, x1 = min([b['bbox'][0] for b in delete_chars]), max([b['bbox'][2] for b in delete_chars])
|
||||
else:
|
||||
logger.debug(f"行内公式替换没有发生,尝试下一行匹配, eqinfo={eqinfo}")
|
||||
return False
|
||||
|
||||
# 删除位于x0, x1这两个中间的span
|
||||
delete_span = []
|
||||
for span in line['spans']:
|
||||
span_box = span['bbox']
|
||||
if x0<=span_box[0] and span_box[2]<=x1:
|
||||
delete_span.append(span)
|
||||
for span in delete_span:
|
||||
line['spans'].remove(span)
|
||||
|
||||
|
||||
equation_span = {
|
||||
"size": 9.962599754333496,
|
||||
"_type": TYPE_INLINE_EQUATION,
|
||||
"flags": 4,
|
||||
"font": TYPE_INLINE_EQUATION,
|
||||
"color": 0,
|
||||
"ascender": 0.9409999847412109,
|
||||
"descender": -0.3050000071525574,
|
||||
"text": "",
|
||||
"origin": [
|
||||
337.1410153102337,
|
||||
216.0205245153934
|
||||
],
|
||||
"bbox": [
|
||||
337.1410153102337,
|
||||
216.0205245153934,
|
||||
390.4496373892022,
|
||||
228.50171037628277
|
||||
]
|
||||
}
|
||||
#equation_span = line['spans'][0].copy()
|
||||
equation_span['text'] = f" ${eqinfo['latex_text']}$ "
|
||||
equation_span['bbox'] = [x0, equation_span['bbox'][1], x1, equation_span['bbox'][3]]
|
||||
equation_span['origin'] = [equation_span['bbox'][0], equation_span['bbox'][1]]
|
||||
equation_span['chars'] = delete_chars
|
||||
equation_span['_type'] = TYPE_INLINE_EQUATION
|
||||
equation_span['_eq_bbox'] = eqinfo['bbox']
|
||||
line['spans'].insert(first_overlap_span_idx+1, equation_span) # 放入公式
|
||||
|
||||
# logger.info(f"==>text is 【{line_txt}】, equation is 【{eqinfo['latex_text']}】")
|
||||
|
||||
# 第一个、和最后一个有overlap的span进行分割,然后插入对应的位置
|
||||
first_span_chars = [char for char in first_overlap_span['chars'] if (char['bbox'][2]+char['bbox'][0])/2<x0]
|
||||
tail_span_chars = [char for char in last_overlap_span['chars'] if (char['bbox'][0]+char['bbox'][2])/2>x1]
|
||||
|
||||
if len(first_span_chars)>0:
|
||||
first_overlap_span['chars'] = first_span_chars
|
||||
first_overlap_span['text'] = ''.join([char['c'] for char in first_span_chars])
|
||||
first_overlap_span['bbox'] = (first_overlap_span['bbox'][0], first_overlap_span['bbox'][1], max([chr['bbox'][2] for chr in first_span_chars]), first_overlap_span['bbox'][3])
|
||||
# first_overlap_span['_type'] = "first"
|
||||
else:
|
||||
# 删掉
|
||||
if first_overlap_span not in delete_span:
|
||||
line['spans'].remove(first_overlap_span)
|
||||
|
||||
|
||||
if len(tail_span_chars)>0:
|
||||
if last_overlap_span==first_overlap_span: # 这个时候应该插入一个新的
|
||||
tail_span_txt = ''.join([char['c'] for char in tail_span_chars])
|
||||
last_span_to_insert = last_overlap_span.copy()
|
||||
last_span_to_insert['chars'] = tail_span_chars
|
||||
last_span_to_insert['text'] = ''.join([char['c'] for char in tail_span_chars])
|
||||
last_span_to_insert['bbox'] = (min([chr['bbox'][0] for chr in tail_span_chars]), last_overlap_span['bbox'][1], last_overlap_span['bbox'][2], last_overlap_span['bbox'][3])
|
||||
# 插入到公式对象之后
|
||||
equation_idx = line['spans'].index(equation_span)
|
||||
line['spans'].insert(equation_idx+1, last_span_to_insert) # 放入公式
|
||||
else: # 直接修改原来的span
|
||||
last_overlap_span['chars'] = tail_span_chars
|
||||
last_overlap_span['text'] = ''.join([char['c'] for char in tail_span_chars])
|
||||
last_overlap_span['bbox'] = (min([chr['bbox'][0] for chr in tail_span_chars]), last_overlap_span['bbox'][1], last_overlap_span['bbox'][2], last_overlap_span['bbox'][3])
|
||||
else:
|
||||
# 删掉
|
||||
if last_overlap_span not in delete_span and last_overlap_span!=first_overlap_span:
|
||||
line['spans'].remove(last_overlap_span)
|
||||
|
||||
remain_txt = ""
|
||||
for span in line['spans']:
|
||||
span_txt = "<span>"
|
||||
for char in span['chars']:
|
||||
span_txt = span_txt + char['c']
|
||||
|
||||
span_txt = span_txt + "</span>"
|
||||
|
||||
remain_txt = remain_txt + span_txt
|
||||
|
||||
# logger.info(f"<== succ replace, text is 【{remain_txt}】, equation is 【{eqinfo['latex_text']}】")
|
||||
|
||||
return True
|
||||
|
||||
|
||||
def replace_eq_blk(eqinfo, text_block):
|
||||
"""替换行内公式"""
|
||||
for line in text_block['lines']:
|
||||
line_bbox = line['bbox']
|
||||
if _is_xin(eqinfo['bbox'], line_bbox) or __y_overlap_ratio(eqinfo['bbox'], line_bbox)>0.6: # 定位到行, 使用y方向重合率是因为有的时候,一个行的宽度会小于公式位置宽度:行很高,公式很窄,
|
||||
replace_succ = replace_line_v2(eqinfo, line)
|
||||
if not replace_succ: # 有的时候,一个pdf的line高度从API里会计算的有问题,因此在行内span级别会替换不成功,这就需要继续重试下一行
|
||||
continue
|
||||
else:
|
||||
break
|
||||
else:
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def replace_inline_equations(inline_equation_bboxes, raw_text_blocks):
|
||||
"""替换行内公式"""
|
||||
for eqinfo in inline_equation_bboxes:
|
||||
eqbox = eqinfo['bbox']
|
||||
for blk in raw_text_blocks:
|
||||
if _is_xin(eqbox, blk['bbox']):
|
||||
if not replace_eq_blk(eqinfo, blk):
|
||||
logger.error(f"行内公式没有替换成功:{eqinfo} ")
|
||||
else:
|
||||
break
|
||||
|
||||
return raw_text_blocks
|
||||
|
||||
def remove_chars_in_text_blocks(text_blocks):
|
||||
"""删除text_blocks里的char"""
|
||||
for blk in text_blocks:
|
||||
for line in blk['lines']:
|
||||
for span in line['spans']:
|
||||
_ = span.pop("chars", "no such key")
|
||||
return text_blocks
|
||||
|
||||
|
||||
def replace_equations_in_textblock(raw_text_blocks, inline_equation_bboxes, interline_equation_bboxes):
|
||||
"""
|
||||
替换行间和和行内公式为latex
|
||||
"""
|
||||
|
||||
raw_text_blocks = remove_text_block_in_interline_equation_bbox(interline_equation_bboxes, raw_text_blocks) # 消除重叠:第一步,在公式内部的
|
||||
raw_text_blocks = remove_text_block_overlap_interline_equation_bbox(interline_equation_bboxes, raw_text_blocks) # 消重,第二步,和公式覆盖的
|
||||
insert_interline_equations_textblock(interline_equation_bboxes, raw_text_blocks)
|
||||
|
||||
raw_text_blocks = replace_inline_equations(inline_equation_bboxes, raw_text_blocks)
|
||||
|
||||
return raw_text_blocks
|
||||
|
||||
|
||||
def draw_block_on_pdf_with_txt_replace_eq_bbox(json_path, pdf_path):
|
||||
"""
|
||||
"""
|
||||
new_pdf = f"{Path(pdf_path).parent}/{Path(pdf_path).stem}.step3-消除行内公式text_block.pdf"
|
||||
with open(json_path, "r", encoding='utf-8') as f:
|
||||
obj = json.loads(f.read())
|
||||
|
||||
if os.path.exists(new_pdf):
|
||||
os.remove(new_pdf)
|
||||
new_doc = fitz.open('')
|
||||
|
||||
doc = fitz.open(pdf_path)
|
||||
new_doc = fitz.open(pdf_path)
|
||||
for i in range(len(new_doc)):
|
||||
page = new_doc[i]
|
||||
inline_equation_bboxes = obj[f"page_{i}"]['inline_equations']
|
||||
interline_equation_bboxes = obj[f"page_{i}"]['interline_equations']
|
||||
raw_text_blocks = obj[f'page_{i}']['preproc_blocks']
|
||||
raw_text_blocks = remove_text_block_in_interline_equation_bbox(interline_equation_bboxes, raw_text_blocks) # 消除重叠:第一步,在公式内部的
|
||||
raw_text_blocks = remove_text_block_overlap_interline_equation_bbox(interline_equation_bboxes, raw_text_blocks) # 消重,第二步,和公式覆盖的
|
||||
insert_interline_equations_textblock(interline_equation_bboxes, raw_text_blocks)
|
||||
raw_text_blocks = replace_inline_equations(inline_equation_bboxes, raw_text_blocks)
|
||||
|
||||
|
||||
# 为了检验公式是否重复,把每一行里,含有公式的span背景改成黄色的
|
||||
color_map = [fitz.pdfcolor['blue'],fitz.pdfcolor['green']]
|
||||
j = 0
|
||||
for blk in raw_text_blocks:
|
||||
for i,line in enumerate(blk['lines']):
|
||||
|
||||
# line_box = line['bbox']
|
||||
# shape = page.new_shape()
|
||||
# shape.draw_rect(line_box)
|
||||
# shape.finish(color=fitz.pdfcolor['red'], fill=color_map[j%2], fill_opacity=0.3)
|
||||
# shape.commit()
|
||||
# j = j+1
|
||||
|
||||
for i, span in enumerate(line['spans']):
|
||||
shape_page = page.new_shape()
|
||||
span_type = span.get('_type')
|
||||
color = fitz.pdfcolor['blue']
|
||||
if span_type=='first':
|
||||
color = fitz.pdfcolor['blue']
|
||||
elif span_type=='tail':
|
||||
color = fitz.pdfcolor['green']
|
||||
elif span_type==TYPE_INLINE_EQUATION:
|
||||
color = fitz.pdfcolor['black']
|
||||
else:
|
||||
color = None
|
||||
|
||||
b = span['bbox']
|
||||
shape_page.draw_rect(b)
|
||||
|
||||
shape_page.finish(color=None, fill=color, fill_opacity=0.3)
|
||||
shape_page.commit()
|
||||
|
||||
new_doc.save(new_pdf)
|
||||
logger.info(f"save ok {new_pdf}")
|
||||
final_json = json.dumps(obj, ensure_ascii=False,indent=2)
|
||||
with open("equations_test/final_json.json", "w") as f:
|
||||
f.write(final_json)
|
||||
|
||||
return new_pdf
|
||||
|
||||
|
||||
if __name__=="__main__":
|
||||
# draw_block_on_pdf_with_txt_replace_eq_bbox(new_json_path, equation_color_pdf)
|
||||
pass
|
||||
@@ -0,0 +1,245 @@
|
||||
|
||||
|
||||
|
||||
import re
|
||||
from libs.boxbase import _is_in_or_part_overlap, _is_part_overlap, _is_in, find_bottom_nearest_text_bbox, find_left_nearest_text_bbox, find_right_nearest_text_bbox, find_top_nearest_text_bbox
|
||||
from loguru import logger
|
||||
|
||||
from libs.textbase import get_text_block_base_info
|
||||
|
||||
def fix_image_vertical(image_bboxes:list, text_blocks:list):
|
||||
"""
|
||||
修正图片的位置
|
||||
如果图片与文字block发生一定重叠(也就是图片切到了一部分文字),那么减少图片边缘,让文字和图片不再重叠。
|
||||
只对垂直方向进行。
|
||||
"""
|
||||
for image_bbox in image_bboxes:
|
||||
for text_block in text_blocks:
|
||||
text_bbox = text_block["bbox"]
|
||||
if _is_part_overlap(text_bbox, image_bbox) and any([text_bbox[0]>=image_bbox[0] and text_bbox[2]<=image_bbox[2], text_bbox[0]<=image_bbox[0] and text_bbox[2]>=image_bbox[2]]):
|
||||
if text_bbox[1] < image_bbox[1]:#在图片上方
|
||||
image_bbox[1] = text_bbox[3]+1
|
||||
elif text_bbox[3]>image_bbox[3]:#在图片下方
|
||||
image_bbox[3] = text_bbox[1]-1
|
||||
|
||||
return image_bboxes
|
||||
|
||||
def __merge_if_common_edge(bbox1, bbox2):
|
||||
x_min_1, y_min_1, x_max_1, y_max_1 = bbox1
|
||||
x_min_2, y_min_2, x_max_2, y_max_2 = bbox2
|
||||
|
||||
# 检查是否有公共的水平边
|
||||
if y_min_1 == y_min_2 or y_max_1 == y_max_2:
|
||||
# 确保一个框的x范围在另一个框的x范围内
|
||||
if max(x_min_1, x_min_2) <= min(x_max_1, x_max_2):
|
||||
return [min(x_min_1, x_min_2), min(y_min_1, y_min_2), max(x_max_1, x_max_2), max(y_max_1, y_max_2)]
|
||||
|
||||
# 检查是否有公共的垂直边
|
||||
if x_min_1 == x_min_2 or x_max_1 == x_max_2:
|
||||
# 确保一个框的y范围在另一个框的y范围内
|
||||
if max(y_min_1, y_min_2) <= min(y_max_1, y_max_2):
|
||||
return [min(x_min_1, x_min_2), min(y_min_1, y_min_2), max(x_max_1, x_max_2), max(y_max_1, y_max_2)]
|
||||
|
||||
# 如果没有公共边
|
||||
return None
|
||||
|
||||
def fix_seperated_image(image_bboxes:list):
|
||||
"""
|
||||
如果2个图片有一个边重叠,那么合并2个图片
|
||||
"""
|
||||
new_images = []
|
||||
droped_img_idx = []
|
||||
|
||||
for i in range(0, len(image_bboxes)):
|
||||
for j in range(i+1, len(image_bboxes)):
|
||||
new_img = __merge_if_common_edge(image_bboxes[i], image_bboxes[j])
|
||||
if new_img is not None:
|
||||
new_images.append(new_img)
|
||||
droped_img_idx.append(i)
|
||||
droped_img_idx.append(j)
|
||||
break
|
||||
|
||||
for i in range(0, len(image_bboxes)):
|
||||
if i not in droped_img_idx:
|
||||
new_images.append(image_bboxes[i])
|
||||
|
||||
return new_images
|
||||
|
||||
|
||||
def __check_img_title_pattern(text):
|
||||
"""
|
||||
检查文本段是否是表格的标题
|
||||
"""
|
||||
patterns = [r"^(fig|figure).*", r"^(scheme).*"]
|
||||
text = text.strip()
|
||||
for pattern in patterns:
|
||||
match = re.match(pattern, text, re.IGNORECASE)
|
||||
if match:
|
||||
return True
|
||||
return False
|
||||
|
||||
def __get_fig_caption_text(text_block):
|
||||
txt = " ".join(span['text'] for line in text_block['lines'] for span in line['spans'])
|
||||
line_cnt = len(text_block['lines'])
|
||||
txt = txt.replace("Ž . ", '')
|
||||
return txt, line_cnt
|
||||
|
||||
|
||||
def __find_and_extend_bottom_caption(text_block, pymu_blocks, image_box):
|
||||
"""
|
||||
继续向下方寻找和图片caption字号,字体,颜色一样的文字框,合并入caption。
|
||||
text_block是已经找到的图片catpion(这个caption可能不全,多行被划分到多个pymu block里了)
|
||||
"""
|
||||
combined_image_caption_text_block = list(text_block.copy()['bbox'])
|
||||
base_font_color, base_font_size, base_font_type = get_text_block_base_info(text_block)
|
||||
while True:
|
||||
tb_add = find_bottom_nearest_text_bbox(pymu_blocks, combined_image_caption_text_block)
|
||||
if not tb_add:
|
||||
break
|
||||
tb_font_color, tb_font_size, tb_font_type = get_text_block_base_info(tb_add)
|
||||
if tb_font_color==base_font_color and tb_font_size==base_font_size and tb_font_type==base_font_type:
|
||||
combined_image_caption_text_block[0] = min(combined_image_caption_text_block[0], tb_add['bbox'][0])
|
||||
combined_image_caption_text_block[2] = max(combined_image_caption_text_block[2], tb_add['bbox'][2])
|
||||
combined_image_caption_text_block[3] = tb_add['bbox'][3]
|
||||
else:
|
||||
break
|
||||
|
||||
image_box[0] = min(image_box[0], combined_image_caption_text_block[0])
|
||||
image_box[1] = min(image_box[1], combined_image_caption_text_block[1])
|
||||
image_box[2] = max(image_box[2], combined_image_caption_text_block[2])
|
||||
image_box[3] = max(image_box[3], combined_image_caption_text_block[3])
|
||||
text_block['_image_caption'] = True
|
||||
|
||||
|
||||
def include_img_title(pymu_blocks, image_bboxes: list):
|
||||
"""
|
||||
向上方和下方寻找符合图片title的文本block,合并到图片里
|
||||
如果图片上下都有fig的情况怎么办?寻找标题距离最近的那个。
|
||||
---
|
||||
增加对左侧和右侧图片标题的寻找
|
||||
"""
|
||||
|
||||
|
||||
for tb in image_bboxes:
|
||||
# 优先找下方的
|
||||
max_find_cnt = 3 # 向上,向下最多找3个就停止
|
||||
temp_box = tb.copy()
|
||||
while max_find_cnt>0:
|
||||
text_block_btn = find_bottom_nearest_text_bbox(pymu_blocks, temp_box)
|
||||
if text_block_btn:
|
||||
txt, line_cnt = __get_fig_caption_text(text_block_btn)
|
||||
if len(txt.strip())>0:
|
||||
if not __check_img_title_pattern(txt) and max_find_cnt>0 and line_cnt<3: # 设置line_cnt<=2目的是为了跳过子标题,或者有时候图片下方文字没有被图片识别模型放入图片里
|
||||
max_find_cnt = max_find_cnt - 1
|
||||
temp_box[3] = text_block_btn['bbox'][3]
|
||||
continue
|
||||
else:
|
||||
break
|
||||
else:
|
||||
temp_box[3] = text_block_btn['bbox'][3] # 宽度不变,扩大
|
||||
max_find_cnt = max_find_cnt - 1
|
||||
else:
|
||||
break
|
||||
|
||||
max_find_cnt = 3 # 向上,向下最多找3个就停止
|
||||
temp_box = tb.copy()
|
||||
while max_find_cnt>0:
|
||||
text_block_top = find_top_nearest_text_bbox(pymu_blocks, temp_box)
|
||||
if text_block_top:
|
||||
txt, line_cnt = __get_fig_caption_text(text_block_top)
|
||||
if len(txt.strip())>0:
|
||||
if not __check_img_title_pattern(txt) and max_find_cnt>0 and line_cnt <3:
|
||||
max_find_cnt = max_find_cnt - 1
|
||||
temp_box[1] = text_block_top['bbox'][1]
|
||||
continue
|
||||
else:
|
||||
break
|
||||
else:
|
||||
b = text_block_top['bbox']
|
||||
temp_box[1] = b[1] # 宽度不变,扩大
|
||||
max_find_cnt = max_find_cnt - 1
|
||||
else:
|
||||
break
|
||||
|
||||
if text_block_btn and text_block_top and text_block_btn.get("_image_caption", False) is False and text_block_top.get("_image_caption", False) is False :
|
||||
btn_text, _ = __get_fig_caption_text(text_block_btn)
|
||||
top_text, _ = __get_fig_caption_text(text_block_top)
|
||||
if __check_img_title_pattern(btn_text) and __check_img_title_pattern(top_text):
|
||||
# 取距离图片最近的
|
||||
btn_text_distance = text_block_btn['bbox'][1] - tb[3]
|
||||
top_text_distance = tb[1] - text_block_top['bbox'][3]
|
||||
if btn_text_distance<top_text_distance: # caption在下方
|
||||
__find_and_extend_bottom_caption(text_block_btn, pymu_blocks, tb)
|
||||
else:
|
||||
text_block = text_block_top
|
||||
tb[0] = min(tb[0], text_block['bbox'][0])
|
||||
tb[1] = min(tb[1], text_block['bbox'][1])
|
||||
tb[2] = max(tb[2], text_block['bbox'][2])
|
||||
tb[3] = max(tb[3], text_block['bbox'][3])
|
||||
text_block_btn['_image_caption'] = True
|
||||
continue
|
||||
|
||||
text_block = text_block_btn # find_bottom_nearest_text_bbox(pymu_blocks, tb)
|
||||
if text_block and text_block.get("_image_caption", False) is False:
|
||||
first_text_line, _ = __get_fig_caption_text(text_block)
|
||||
if __check_img_title_pattern(first_text_line):
|
||||
# 发现特征之后,继续向相同方向寻找(想同颜色,想同大小,想同字体)的textblock
|
||||
__find_and_extend_bottom_caption(text_block, pymu_blocks, tb)
|
||||
continue
|
||||
|
||||
text_block = text_block_top # find_top_nearest_text_bbox(pymu_blocks, tb)
|
||||
if text_block and text_block.get("_image_caption", False) is False:
|
||||
first_text_line, _ = __get_fig_caption_text(text_block)
|
||||
if __check_img_title_pattern(first_text_line):
|
||||
tb[0] = min(tb[0], text_block['bbox'][0])
|
||||
tb[1] = min(tb[1], text_block['bbox'][1])
|
||||
tb[2] = max(tb[2], text_block['bbox'][2])
|
||||
tb[3] = max(tb[3], text_block['bbox'][3])
|
||||
text_block['_image_caption'] = True
|
||||
continue
|
||||
|
||||
"""向左、向右寻找,暂时只寻找一次"""
|
||||
left_text_block = find_left_nearest_text_bbox(pymu_blocks, tb)
|
||||
if left_text_block and left_text_block.get("_image_caption", False) is False:
|
||||
first_text_line, _ = __get_fig_caption_text(left_text_block)
|
||||
if __check_img_title_pattern(first_text_line):
|
||||
tb[0] = min(tb[0], left_text_block['bbox'][0])
|
||||
tb[1] = min(tb[1], left_text_block['bbox'][1])
|
||||
tb[2] = max(tb[2], left_text_block['bbox'][2])
|
||||
tb[3] = max(tb[3], left_text_block['bbox'][3])
|
||||
left_text_block['_image_caption'] = True
|
||||
continue
|
||||
|
||||
right_text_block = find_right_nearest_text_bbox(pymu_blocks, tb)
|
||||
if right_text_block and right_text_block.get("_image_caption", False) is False:
|
||||
first_text_line, _ = __get_fig_caption_text(right_text_block)
|
||||
if __check_img_title_pattern(first_text_line):
|
||||
tb[0] = min(tb[0], right_text_block['bbox'][0])
|
||||
tb[1] = min(tb[1], right_text_block['bbox'][1])
|
||||
tb[2] = max(tb[2], right_text_block['bbox'][2])
|
||||
tb[3] = max(tb[3], right_text_block['bbox'][3])
|
||||
right_text_block['_image_caption'] = True
|
||||
continue
|
||||
|
||||
return image_bboxes
|
||||
|
||||
|
||||
def combine_images(image_bboxes:list):
|
||||
"""
|
||||
合并图片,如果图片有重叠,那么合并
|
||||
"""
|
||||
new_images = []
|
||||
droped_img_idx = []
|
||||
|
||||
for i in range(0, len(image_bboxes)):
|
||||
for j in range(i+1, len(image_bboxes)):
|
||||
if j not in droped_img_idx and _is_in_or_part_overlap(image_bboxes[i], image_bboxes[j]):
|
||||
# 合并
|
||||
image_bboxes[i][0], image_bboxes[i][1],image_bboxes[i][2],image_bboxes[i][3] = min(image_bboxes[i][0], image_bboxes[j][0]), min(image_bboxes[i][1], image_bboxes[j][1]), max(image_bboxes[i][2], image_bboxes[j][2]), max(image_bboxes[i][3], image_bboxes[j][3])
|
||||
droped_img_idx.append(j)
|
||||
|
||||
for i in range(0, len(image_bboxes)):
|
||||
if i not in droped_img_idx:
|
||||
new_images.append(image_bboxes[i])
|
||||
|
||||
return new_images
|
||||
@@ -0,0 +1,23 @@
|
||||
import collections
|
||||
|
||||
|
||||
def get_main_text_font(pdf_docs):
|
||||
font_names = collections.Counter()
|
||||
for page in pdf_docs:
|
||||
blocks = page.get_text('dict')['blocks']
|
||||
if blocks is not None:
|
||||
for block in blocks:
|
||||
lines = block.get('lines')
|
||||
if lines is not None:
|
||||
for line in lines:
|
||||
span_font = [(span['font'], len(span['text'])) for span in line['spans'] if
|
||||
'font' in span and len(span['text']) > 0]
|
||||
if span_font:
|
||||
# main_text_font应该用基于字数最多的字体而不是span级别的统计
|
||||
# font_names.append(font_name for font_name in span_font)
|
||||
# block_fonts.append(font_name for font_name in span_font)
|
||||
for font, count in span_font:
|
||||
font_names[font] += count
|
||||
main_text_font = font_names.most_common(1)[0][0]
|
||||
return main_text_font
|
||||
|
||||
@@ -0,0 +1,74 @@
|
||||
from libs.commons import fitz
|
||||
from libs.boxbase import _is_in, _is_in_or_part_overlap
|
||||
from libs.drop_reason import DropReason
|
||||
|
||||
|
||||
def __area(box):
|
||||
return (box[2] - box[0]) * (box[3] - box[1])
|
||||
|
||||
def __is_contain_color_background_rect(page:fitz.Page, text_blocks, image_bboxes) -> bool:
|
||||
"""
|
||||
检查page是包含有颜色背景的矩形
|
||||
"""
|
||||
color_bg_rect = []
|
||||
p_width, p_height = page.rect.width, page.rect.height
|
||||
|
||||
# 先找到最大的带背景矩形
|
||||
blocks = page.get_cdrawings()
|
||||
for block in blocks:
|
||||
|
||||
if 'fill' in block and block['fill']: # 过滤掉透明的
|
||||
fill = list(block['fill'])
|
||||
fill[0], fill[1], fill[2] = int(fill[0]), int(fill[1]), int(fill[2])
|
||||
if fill==(1.0,1.0,1.0):
|
||||
continue
|
||||
rect = block['rect']
|
||||
# 过滤掉特别小的矩形
|
||||
if __area(rect) < 10*10:
|
||||
continue
|
||||
# 为了防止是svg图片上的色块,这里过滤掉这类
|
||||
|
||||
if any([_is_in_or_part_overlap(rect, img_bbox) for img_bbox in image_bboxes]):
|
||||
continue
|
||||
color_bg_rect.append(rect)
|
||||
|
||||
# 找到最大的背景矩形
|
||||
if len(color_bg_rect) > 0:
|
||||
max_rect = max(color_bg_rect, key=lambda x:__area(x))
|
||||
max_rect_int = (int(max_rect[0]), int(max_rect[1]), int(max_rect[2]), int(max_rect[3]))
|
||||
# 判断最大的背景矩形是否包含超过3行文字,或者50个字 TODO
|
||||
if max_rect[2]-max_rect[0] > 0.2*p_width and max_rect[3]-max_rect[1] > 0.1*p_height:#宽度符合
|
||||
#看是否有文本块落入到这个矩形中
|
||||
for text_block in text_blocks:
|
||||
box = text_block['bbox']
|
||||
box_int = (int(box[0]), int(box[1]), int(box[2]), int(box[3]))
|
||||
if _is_in(box_int, max_rect_int):
|
||||
return True
|
||||
|
||||
return False
|
||||
|
||||
|
||||
def __is_table_overlap_text_block(text_blocks, table_bbox):
|
||||
"""
|
||||
检查table_bbox是否覆盖了text_blocks里的文本块
|
||||
TODO
|
||||
"""
|
||||
for text_block in text_blocks:
|
||||
box = text_block['bbox']
|
||||
if _is_in_or_part_overlap(table_bbox, box):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def pdf_filter(page:fitz.Page, text_blocks, table_bboxes, image_bboxes) -> tuple:
|
||||
"""
|
||||
return:(True|False, err_msg)
|
||||
True, 如果pdf符合要求
|
||||
False, 如果pdf不符合要求
|
||||
|
||||
"""
|
||||
if __is_contain_color_background_rect(page, text_blocks, image_bboxes):
|
||||
return False, {"need_drop": True, "drop_reason": DropReason.COLOR_BACKGROUND_TEXT_BOX}
|
||||
|
||||
|
||||
return True, None
|
||||
@@ -0,0 +1,79 @@
|
||||
from libs.boxbase import _is_in, _is_in_or_part_overlap, calculate_overlap_area_2_minbox_area_ratio
|
||||
from loguru import logger
|
||||
|
||||
from libs.drop_tag import COLOR_BG_HEADER_TXT_BLOCK
|
||||
|
||||
|
||||
def __area(box):
|
||||
return (box[2] - box[0]) * (box[3] - box[1])
|
||||
|
||||
|
||||
def rectangle_position_determination(rect, p_width):
|
||||
"""
|
||||
判断矩形是否在页面中轴线附近。
|
||||
|
||||
Args:
|
||||
rect (list): 矩形坐标,格式为[x1, y1, x2, y2]。
|
||||
p_width (int): 页面宽度。
|
||||
|
||||
Returns:
|
||||
bool: 若矩形在页面中轴线附近则返回True,否则返回False。
|
||||
"""
|
||||
# 页面中轴线x坐标
|
||||
x_axis = p_width / 2
|
||||
# 矩形是否跨越中轴线
|
||||
is_span = rect[0] < x_axis and rect[2] > x_axis
|
||||
if is_span:
|
||||
return True
|
||||
else:
|
||||
# 矩形与中轴线的距离,只算近的那一边
|
||||
distance = rect[0] - x_axis if rect[0] > x_axis else x_axis - rect[2]
|
||||
# 判断矩形与中轴线的距离是否小于页面宽度的20%
|
||||
if distance < p_width * 0.2:
|
||||
return True
|
||||
else:
|
||||
return False
|
||||
|
||||
def remove_colored_strip_textblock(remain_text_blocks, page):
|
||||
"""
|
||||
根据页面中特定颜色和大小过滤文本块,将符合条件的文本块从remain_text_blocks中移除,并返回移除的文本块列表colored_strip_textblock。
|
||||
|
||||
Args:
|
||||
remain_text_blocks (list): 剩余文本块列表。
|
||||
page (Page): 页面对象。
|
||||
|
||||
Returns:
|
||||
tuple: 剩余文本块列表和移除的文本块列表。
|
||||
"""
|
||||
colored_strip_textblocks = [] # 先构造一个空的返回
|
||||
if len(remain_text_blocks) > 0:
|
||||
p_width, p_height = page.rect.width, page.rect.height
|
||||
blocks = page.get_cdrawings()
|
||||
colored_strip_bg_rect = []
|
||||
for block in blocks:
|
||||
is_filled = 'fill' in block and block['fill'] and block['fill'] != (1.0, 1.0, 1.0) # 过滤掉透明的
|
||||
rect = block['rect']
|
||||
area_is_large_enough = __area(rect) > 100 # 过滤掉特别小的矩形
|
||||
rectangle_position_determination_result = rectangle_position_determination(rect, p_width)
|
||||
in_upper_half_page = rect[3] < p_height * 0.3 # 找到位于页面上半部分的矩形,下边界小于页面高度的30%
|
||||
aspect_ratio_exceeds_4 = (rect[2] - rect[0]) > (rect[3] - rect[1]) * 4 # 找到长宽比超过4的矩形
|
||||
|
||||
if is_filled and area_is_large_enough and rectangle_position_determination_result and in_upper_half_page and aspect_ratio_exceeds_4:
|
||||
colored_strip_bg_rect.append(rect)
|
||||
|
||||
if len(colored_strip_bg_rect) > 0:
|
||||
for colored_strip_block_bbox in colored_strip_bg_rect:
|
||||
for text_block in remain_text_blocks:
|
||||
text_bbox = text_block['bbox']
|
||||
if _is_in(text_bbox, colored_strip_block_bbox) or (_is_in_or_part_overlap(text_bbox, colored_strip_block_bbox) and calculate_overlap_area_2_minbox_area_ratio(text_bbox, colored_strip_block_bbox) > 0.6):
|
||||
logger.info(f'remove_colored_strip_textblock: {text_bbox}, {colored_strip_block_bbox}')
|
||||
text_block['tag'] = COLOR_BG_HEADER_TXT_BLOCK
|
||||
colored_strip_textblocks.append(text_block)
|
||||
|
||||
if len(colored_strip_textblocks) > 0:
|
||||
for colored_strip_textblock in colored_strip_textblocks:
|
||||
if colored_strip_textblock in remain_text_blocks:
|
||||
remain_text_blocks.remove(colored_strip_textblock)
|
||||
|
||||
return remain_text_blocks, colored_strip_textblocks
|
||||
|
||||
@@ -0,0 +1,189 @@
|
||||
|
||||
import json
|
||||
import math
|
||||
|
||||
from libs.boxbase import is_vbox_on_side
|
||||
|
||||
|
||||
def detect_non_horizontal_texts(result_dict):
|
||||
"""
|
||||
This function detects watermarks and vertical margin notes in the document.
|
||||
|
||||
Watermarks are identified by finding blocks with the same coordinates and frequently occurring identical texts across multiple pages.
|
||||
If these conditions are met, the blocks are highly likely to be watermarks, as opposed to headers or footers, which can change from page to page.
|
||||
If the direction of these blocks is not horizontal, they are definitely considered to be watermarks.
|
||||
|
||||
Vertical margin notes are identified by finding blocks with the same coordinates and frequently occurring identical texts across multiple pages.
|
||||
If these conditions are met, the blocks are highly likely to be vertical margin notes, which typically appear on the left and right sides of the page.
|
||||
If the direction of these blocks is vertical, they are definitely considered to be vertical margin notes.
|
||||
|
||||
|
||||
Parameters
|
||||
----------
|
||||
result_dict : dict
|
||||
The result dictionary.
|
||||
|
||||
Returns
|
||||
-------
|
||||
result_dict : dict
|
||||
The updated result dictionary.
|
||||
"""
|
||||
# Dictionary to store information about potential watermarks
|
||||
potential_watermarks = {}
|
||||
potential_margin_notes = {}
|
||||
|
||||
for page_id, page_content in result_dict.items():
|
||||
if page_id.startswith("page_"):
|
||||
for block_id, block_data in page_content.items():
|
||||
if block_id.startswith("block_"):
|
||||
if "dir" in block_data:
|
||||
coordinates_text = (block_data["bbox"], block_data["text"]) # Tuple of coordinates and text
|
||||
|
||||
angle = math.atan2(block_data["dir"][1], block_data["dir"][0])
|
||||
angle = abs(math.degrees(angle))
|
||||
|
||||
if angle > 5 and angle < 85: # Check if direction is watermarks
|
||||
if coordinates_text in potential_watermarks:
|
||||
potential_watermarks[coordinates_text] += 1
|
||||
else:
|
||||
potential_watermarks[coordinates_text] = 1
|
||||
|
||||
if angle > 85 and angle < 105: # Check if direction is vertical
|
||||
if coordinates_text in potential_margin_notes:
|
||||
potential_margin_notes[coordinates_text] += 1 # Increment count
|
||||
else:
|
||||
potential_margin_notes[coordinates_text] = 1 # Initialize count
|
||||
|
||||
# Identify watermarks by finding entries with counts higher than a threshold (e.g., appearing on more than half of the pages)
|
||||
watermark_threshold = len(result_dict) // 2
|
||||
watermarks = {k: v for k, v in potential_watermarks.items() if v > watermark_threshold}
|
||||
|
||||
# Identify margin notes by finding entries with counts higher than a threshold (e.g., appearing on more than half of the pages)
|
||||
margin_note_threshold = len(result_dict) // 2
|
||||
margin_notes = {k: v for k, v in potential_margin_notes.items() if v > margin_note_threshold}
|
||||
|
||||
# Add watermark information to the result dictionary
|
||||
for page_id, blocks in result_dict.items():
|
||||
if page_id.startswith("page_"):
|
||||
for block_id, block_data in blocks.items():
|
||||
coordinates_text = (block_data["bbox"], block_data["text"])
|
||||
if coordinates_text in watermarks:
|
||||
block_data["is_watermark"] = 1
|
||||
else:
|
||||
block_data["is_watermark"] = 0
|
||||
|
||||
if coordinates_text in margin_notes:
|
||||
block_data["is_vertical_margin_note"] = 1
|
||||
else:
|
||||
block_data["is_vertical_margin_note"] = 0
|
||||
|
||||
return result_dict
|
||||
|
||||
|
||||
"""
|
||||
1. 当一个block里全部文字都不是dir=(1,0),这个block整体去掉
|
||||
2. 当一个block里全部文字都是dir=(1,0),但是每行只有一个字,这个block整体去掉。这个block必须出现在页面的四周,否则不去掉
|
||||
"""
|
||||
import string, re
|
||||
|
||||
def __is_a_word(sentence):
|
||||
# 如果输入是中文并且长度为1,则返回True
|
||||
if re.fullmatch(r'[\u4e00-\u9fa5]', sentence):
|
||||
return True
|
||||
# 判断是否为单个英文单词或字符(包括ASCII标点)
|
||||
elif re.fullmatch(r'[a-zA-Z0-9]+', sentence) and len(sentence) <=2:
|
||||
return True
|
||||
else:
|
||||
return False
|
||||
|
||||
|
||||
def __get_text_color(num):
|
||||
"""获取字体的颜色RGB值"""
|
||||
blue = num & 255
|
||||
green = (num >> 8) & 255
|
||||
red = (num >> 16) & 255
|
||||
return red, green, blue
|
||||
|
||||
|
||||
def __is_empty_side_box(text_block):
|
||||
"""
|
||||
是否是边缘上的空白没有任何内容的block
|
||||
"""
|
||||
for line in text_block['lines']:
|
||||
for span in line['spans']:
|
||||
font_color = span['color']
|
||||
r,g,b = __get_text_color(font_color)
|
||||
if len(span['text'].strip())>0 and (r,g,b)!=(255,255,255):
|
||||
return False
|
||||
|
||||
return True
|
||||
|
||||
|
||||
def remove_rotate_side_textblock(pymu_text_block, page_width, page_height):
|
||||
"""
|
||||
返回删除了垂直,水印,旋转的textblock
|
||||
删除的内容打上tag返回
|
||||
"""
|
||||
removed_text_block = []
|
||||
|
||||
for i, block in enumerate(pymu_text_block): # 格式参考test/assets/papre/pymu_textblocks.json
|
||||
lines = block['lines']
|
||||
block_bbox = block['bbox']
|
||||
if not is_vbox_on_side(block_bbox, page_width, page_height, 0.2): # 保证这些box必须在页面的两边
|
||||
continue
|
||||
|
||||
if all([__is_a_word(line['spans'][0]["text"]) for line in lines if len(line['spans'])>0]) and len(lines)>1 and all([len(line['spans'])==1 for line in lines]):
|
||||
is_box_valign = (len(set([int(line['spans'][0]['bbox'][0] ) for line in lines if len(line['spans'])>0]))==1) and (len([int(line['spans'][0]['bbox'][0] ) for line in lines if len(line['spans'])>0])>1) # 测试bbox在垂直方向是不是x0都相等,也就是在垂直方向排列.同时必须大于等于2个字
|
||||
|
||||
if is_box_valign:
|
||||
block['tag'] = "vertical-text"
|
||||
removed_text_block.append(block)
|
||||
continue
|
||||
|
||||
for line in lines:
|
||||
if line['dir']!=(1,0):
|
||||
block['tag'] = "rotate"
|
||||
removed_text_block.append(block) # 只要有一个line不是dir=(1,0),就把整个block都删掉
|
||||
break
|
||||
|
||||
for block in removed_text_block:
|
||||
pymu_text_block.remove(block)
|
||||
|
||||
return pymu_text_block, removed_text_block
|
||||
|
||||
def get_side_boundry(rotate_bbox, page_width, page_height):
|
||||
"""
|
||||
根据rotate_bbox,返回页面的左右正文边界
|
||||
"""
|
||||
left_x = 0
|
||||
right_x = page_width
|
||||
for x in rotate_bbox:
|
||||
box = x['bbox']
|
||||
if box[2]<page_width/2:
|
||||
left_x = max(left_x, box[2])
|
||||
else:
|
||||
right_x = min(right_x, box[0])
|
||||
|
||||
return left_x+1, right_x-1
|
||||
|
||||
|
||||
def remove_side_blank_block(pymu_text_block, page_width, page_height):
|
||||
"""
|
||||
删除页面两侧的空白block
|
||||
"""
|
||||
removed_text_block = []
|
||||
|
||||
for i, block in enumerate(pymu_text_block): # 格式参考test/assets/papre/pymu_textblocks.json
|
||||
block_bbox = block['bbox']
|
||||
if not is_vbox_on_side(block_bbox, page_width, page_height, 0.2): # 保证这些box必须在页面的两边
|
||||
continue
|
||||
|
||||
if __is_empty_side_box(block):
|
||||
block['tag'] = "empty-side-block"
|
||||
removed_text_block.append(block)
|
||||
continue
|
||||
|
||||
for block in removed_text_block:
|
||||
pymu_text_block.remove(block)
|
||||
|
||||
return pymu_text_block, removed_text_block
|
||||
@@ -0,0 +1,160 @@
|
||||
|
||||
"""
|
||||
从pdf里提取出来api给出的bbox,然后根据重叠情况做出取舍
|
||||
1. 首先去掉出现在图片上的bbox,图片包括表格和图片
|
||||
2. 然后去掉出现在文字blcok上的图片bbox
|
||||
"""
|
||||
|
||||
from libs.boxbase import _is_in, _is_in_or_part_overlap, _is_left_overlap, calculate_iou, calculate_overlap_area_2_minbox_area_ratio
|
||||
|
||||
|
||||
def resolve_bbox_overlap_conflict(images:list, tables:list, interline_equations:list, inline_equations:list, text_raw_blocks:list):
|
||||
"""
|
||||
text_raw_blocks结构是从pymupdf里直接取到的结构,具体样例参考test/assets/papre/pymu_textblocks.json
|
||||
当下采用一种粗暴的方式:
|
||||
1. 去掉图片上的公式
|
||||
2. 去掉table上的公式
|
||||
2. 图片和文字block部分重叠,首先丢弃图片
|
||||
3. 图片和图片重叠,修改图片的bbox,使得图片不重叠(暂时没这么做,先把图片都扔掉)
|
||||
4. 去掉文字bbox里位于图片、表格上的文字(一定要完全在图、表内部)
|
||||
5. 去掉表格上的文字
|
||||
"""
|
||||
text_block_removed = []
|
||||
images_backup = []
|
||||
|
||||
# 去掉位于图片上的文字block
|
||||
for image_box in images:
|
||||
for text_block in text_raw_blocks:
|
||||
text_bbox = text_block["bbox"]
|
||||
if _is_in(text_bbox, image_box):
|
||||
text_block['tag'] = "on-image"
|
||||
text_block_removed.append(text_block)
|
||||
# 去掉table上的文字block
|
||||
for table_box in tables:
|
||||
for text_block in text_raw_blocks:
|
||||
text_bbox = text_block["bbox"]
|
||||
if _is_in(text_bbox, table_box):
|
||||
text_block['tag'] = "on-table"
|
||||
text_block_removed.append(text_block)
|
||||
|
||||
for text_block in text_block_removed:
|
||||
if text_block in text_raw_blocks:
|
||||
text_raw_blocks.remove(text_block)
|
||||
|
||||
# 第一步去掉在图片上出现的公式box
|
||||
temp = []
|
||||
for image_box in images:
|
||||
for eq1 in interline_equations:
|
||||
if _is_in_or_part_overlap(image_box, eq1[:4]):
|
||||
temp.append(eq1)
|
||||
for eq2 in inline_equations:
|
||||
if _is_in_or_part_overlap(image_box, eq2[:4]):
|
||||
temp.append(eq2)
|
||||
|
||||
for eq in temp:
|
||||
if eq in interline_equations:
|
||||
interline_equations.remove(eq)
|
||||
if eq in inline_equations:
|
||||
inline_equations.remove(eq)
|
||||
|
||||
# 第二步去掉在表格上出现的公式box
|
||||
temp = []
|
||||
for table_box in tables:
|
||||
for eq1 in interline_equations:
|
||||
if _is_in_or_part_overlap(table_box, eq1[:4]):
|
||||
temp.append(eq1)
|
||||
for eq2 in inline_equations:
|
||||
if _is_in_or_part_overlap(table_box, eq2[:4]):
|
||||
temp.append(eq2)
|
||||
|
||||
for eq in temp:
|
||||
if eq in interline_equations:
|
||||
interline_equations.remove(eq)
|
||||
if eq in inline_equations:
|
||||
inline_equations.remove(eq)
|
||||
|
||||
# 图片和文字重叠,丢掉图片
|
||||
for image_box in images:
|
||||
for text_block in text_raw_blocks:
|
||||
text_bbox = text_block["bbox"]
|
||||
if _is_in_or_part_overlap(image_box, text_bbox):
|
||||
images_backup.append(image_box)
|
||||
break
|
||||
for image_box in images_backup:
|
||||
images.remove(image_box)
|
||||
|
||||
# 图片和图片重叠,两张都暂时不参与版面计算
|
||||
images_dup_index = []
|
||||
for i in range(len(images)):
|
||||
for j in range(i+1, len(images)):
|
||||
if _is_in_or_part_overlap(images[i], images[j]):
|
||||
images_dup_index.append(i)
|
||||
images_dup_index.append(j)
|
||||
|
||||
dup_idx = set(images_dup_index)
|
||||
for img_id in dup_idx:
|
||||
images_backup.append(images[img_id])
|
||||
images[img_id] = None
|
||||
|
||||
images = [img for img in images if img is not None]
|
||||
|
||||
# 如果行间公式和文字block重叠,放到临时的数据里,防止这些文字box影响到layout计算。通过计算IOU合并行间公式和文字block
|
||||
# 对于这样的文本块删除,然后保留行间公式的大小不变。
|
||||
# 当计算完毕layout,这部分再合并回来
|
||||
text_block_removed_2 = []
|
||||
# for text_block in text_raw_blocks:
|
||||
# text_bbox = text_block["bbox"]
|
||||
# for eq in interline_equations:
|
||||
# ratio = calculate_overlap_area_2_minbox_area_ratio(text_bbox, eq[:4])
|
||||
# if ratio>0.05:
|
||||
# text_block['tag'] = "belong-to-interline-equation"
|
||||
# text_block_removed_2.append(text_block)
|
||||
# break
|
||||
|
||||
# for tb in text_block_removed_2:
|
||||
# if tb in text_raw_blocks:
|
||||
# text_raw_blocks.remove(tb)
|
||||
|
||||
# text_block_removed = text_block_removed + text_block_removed_2
|
||||
|
||||
return images, tables, interline_equations, inline_equations, text_raw_blocks, text_block_removed, images_backup, text_block_removed_2
|
||||
|
||||
|
||||
def check_text_block_horizontal_overlap(text_blocks:list, header, footer) -> bool:
|
||||
"""
|
||||
检查文本block之间的水平重叠情况,这种情况如果发生,那么这个pdf就不再继续处理了。
|
||||
因为这种情况大概率发生了公式没有被检测出来。
|
||||
|
||||
"""
|
||||
if len(text_blocks)==0:
|
||||
return False
|
||||
|
||||
page_min_y = 0
|
||||
page_max_y = max(yy['bbox'][3] for yy in text_blocks)
|
||||
|
||||
def __max_y(lst:list):
|
||||
if len(lst)>0:
|
||||
return max([item[1] for item in lst])
|
||||
return page_min_y
|
||||
|
||||
def __min_y(lst:list):
|
||||
if len(lst)>0:
|
||||
return min([item[3] for item in lst])
|
||||
return page_max_y
|
||||
|
||||
clip_y0 = __max_y(header)
|
||||
clip_y1 = __min_y(footer)
|
||||
|
||||
txt_bboxes = []
|
||||
for text_block in text_blocks:
|
||||
bbox = text_block["bbox"]
|
||||
if bbox[1]>=clip_y0 and bbox[3]<=clip_y1:
|
||||
txt_bboxes.append(bbox)
|
||||
|
||||
for i in range(len(txt_bboxes)):
|
||||
for j in range(i+1, len(txt_bboxes)):
|
||||
if _is_left_overlap(txt_bboxes[i], txt_bboxes[j]) or _is_left_overlap(txt_bboxes[j], txt_bboxes[i]):
|
||||
return True
|
||||
|
||||
return False
|
||||
|
||||
@@ -0,0 +1,12 @@
|
||||
|
||||
"""
|
||||
统计处需要跨页、全局性的数据
|
||||
- 统计出字号从大到小
|
||||
- 正文区域占比最高的前5
|
||||
- 正文平均行间距
|
||||
- 正文平均字间距
|
||||
- 正文平均字符宽度
|
||||
- 正文平均字符高度
|
||||
|
||||
"""
|
||||
|
||||
@@ -0,0 +1,274 @@
|
||||
import os
|
||||
import collections # 统计库
|
||||
import re # 正则
|
||||
from libs.commons import fitz # pyMuPDF库
|
||||
import json
|
||||
import re
|
||||
|
||||
from libs.boxbase import _is_in_or_part_overlap, _is_part_overlap, find_bottom_nearest_text_bbox, find_left_nearest_text_bbox, find_right_nearest_text_bbox, find_top_nearest_text_bbox # json
|
||||
|
||||
|
||||
## version 2
|
||||
def get_merged_line(page):
|
||||
"""
|
||||
这个函数是为了从pymuPDF中提取出的矢量里筛出水平的横线,并且将断开的线段进行了合并。
|
||||
:param page :fitz读取的当前页的内容
|
||||
"""
|
||||
drawings_bbox = []
|
||||
drawings_line = []
|
||||
drawings = page.get_drawings() # 提取所有的矢量
|
||||
for p in drawings:
|
||||
drawings_bbox.append(p["rect"].irect) # (L, U, R, D)
|
||||
|
||||
lines = []
|
||||
for L, U, R, D in drawings_bbox:
|
||||
if abs(D - U) <= 3: # 筛出水平的横线
|
||||
lines.append((L, U, R, D))
|
||||
U_groups = []
|
||||
visited = [False for _ in range(len(lines))]
|
||||
for i, (L1, U1, R1, D1) in enumerate(lines):
|
||||
if visited[i] == True:
|
||||
continue
|
||||
tmp_g = [(L1, U1, R1, D1)]
|
||||
for j, (L2, U2, R2, D2) in enumerate(lines):
|
||||
if i == j:
|
||||
continue
|
||||
if visited[j] == True:
|
||||
continue
|
||||
if max(U1, D1, U2, D2) - min(U1, D1, U2, D2) <= 5: # 把高度一致的线放进一个group
|
||||
tmp_g.append((L2, U2, R2, D2))
|
||||
visited[j] = True
|
||||
U_groups.append(tmp_g)
|
||||
|
||||
res = []
|
||||
for group in U_groups:
|
||||
group.sort(key = lambda LURD: (LURD[0], LURD[2]))
|
||||
LL, UU, RR, DD = group[0]
|
||||
for i, (L1, U1, R1, D1) in enumerate(group):
|
||||
if (L1 - RR) >= 5:
|
||||
cur_line = (LL, UU, RR, DD)
|
||||
res.append(cur_line)
|
||||
LL = L1
|
||||
else:
|
||||
RR = max(RR, R1)
|
||||
cur_line = (LL, UU, RR, DD)
|
||||
res.append(cur_line)
|
||||
return res
|
||||
|
||||
def fix_tables(page: fitz.Page, table_bboxes: list, include_table_title: bool, scan_line_num: int):
|
||||
"""
|
||||
:param page :fitz读取的当前页的内容
|
||||
:param table_bboxes: list类型,每一个元素是一个元祖 (L, U, R, D)
|
||||
:param include_table_title: 是否将表格的标题也圈进来
|
||||
:param scan_line_num: 在与表格框临近的上下几个文本框里扫描搜索标题
|
||||
"""
|
||||
|
||||
drawings_lines = get_merged_line(page)
|
||||
fix_table_bboxes = []
|
||||
|
||||
for table in table_bboxes:
|
||||
(L, U, R, D) = table
|
||||
fix_table_L = []
|
||||
fix_table_U = []
|
||||
fix_table_R = []
|
||||
fix_table_D = []
|
||||
width = R - L
|
||||
width_range = width * 0.1 # 只看距离表格整体宽度10%之内偏差的线
|
||||
height = D - U
|
||||
height_range = height * 0.1 # 只看距离表格整体高度10%之内偏差的线
|
||||
for line in drawings_lines:
|
||||
if (L - width_range) <= line[0] <= (L + width_range) and (R - width_range) <= line[2] <= (R + width_range): # 相近的宽度
|
||||
if (U - height_range) < line[1] < (U + height_range): # 上边界,在一定的高度范围内
|
||||
fix_table_U.append(line[1])
|
||||
fix_table_L.append(line[0])
|
||||
fix_table_R.append(line[2])
|
||||
elif (D - height_range) < line[1] < (D + height_range): # 下边界,在一定的高度范围内
|
||||
fix_table_D.append(line[1])
|
||||
fix_table_L.append(line[0])
|
||||
fix_table_R.append(line[2])
|
||||
|
||||
if fix_table_U:
|
||||
U = min(fix_table_U)
|
||||
if fix_table_D:
|
||||
D = max(fix_table_D)
|
||||
if fix_table_L:
|
||||
L = min(fix_table_L)
|
||||
if fix_table_R:
|
||||
R = max(fix_table_R)
|
||||
|
||||
if include_table_title: # 需要将表格标题包括
|
||||
text_blocks = page.get_text("dict", flags=fitz.TEXTFLAGS_TEXT)["blocks"] # 所有的text的block
|
||||
incolumn_text_blocks = [block for block in text_blocks if not ((block['bbox'][0] < L and block['bbox'][2] < L) or (block['bbox'][0] > R and block['bbox'][2] > R))] # 将与表格完全没有任何遮挡的文字筛除掉(比如另一栏的文字)
|
||||
upper_text_blocks = [block for block in incolumn_text_blocks if (U - block['bbox'][3]) > 0] # 将在表格线以上的text block筛选出来
|
||||
sorted_filtered_text_blocks = sorted(upper_text_blocks, key=lambda x: (U - x['bbox'][3], x['bbox'][0])) # 按照text block的下边界距离表格上边界的距离升序排序,如果是同一个高度,则先左再右
|
||||
|
||||
for idx in range(scan_line_num):
|
||||
if idx+1 <= len(sorted_filtered_text_blocks):
|
||||
line_temp = sorted_filtered_text_blocks[idx]['lines']
|
||||
if line_temp:
|
||||
text = line_temp[0]['spans'][0]['text'] # 提取出第一个span里的text内容
|
||||
check_en = re.match('Table', text) # 检查是否有Table开头的(英文)
|
||||
check_ch = re.match('表', text) # 检查是否有Table开头的(中文)
|
||||
if check_en or check_ch:
|
||||
if sorted_filtered_text_blocks[idx]['bbox'][1] < D: # 以防出现负的bbox
|
||||
U = sorted_filtered_text_blocks[idx]['bbox'][1]
|
||||
|
||||
fix_table_bboxes.append([L-2, U-2, R+2, D+2])
|
||||
|
||||
return fix_table_bboxes
|
||||
|
||||
def __check_table_title_pattern(text):
|
||||
"""
|
||||
检查文本段是否是表格的标题
|
||||
"""
|
||||
patterns = [r'^table\s\d+']
|
||||
|
||||
for pattern in patterns:
|
||||
match = re.match(pattern, text, re.IGNORECASE)
|
||||
if match:
|
||||
return True
|
||||
else:
|
||||
return False
|
||||
|
||||
|
||||
def fix_table_text_block(pymu_blocks, table_bboxes: list):
|
||||
"""
|
||||
调整table, 如果table和上下的text block有相交区域,则将table的上下边界调整到text block的上下边界
|
||||
例如 tmp/unittest/unittest_pdf/纯2列_ViLT_6_文字 表格.pdf
|
||||
"""
|
||||
for tb in table_bboxes:
|
||||
(L, U, R, D) = tb
|
||||
for block in pymu_blocks:
|
||||
if _is_in_or_part_overlap((L, U, R, D), block['bbox']):
|
||||
txt = " ".join(span['text'] for line in block['lines'] for span in line['spans'])
|
||||
if not __check_table_title_pattern(txt) and block.get("_table", False) is False: # 如果是table的title,那么不调整。因为下一步会统一调整,如果这里进行了调整,后面的调整会造成调整到其他table的title上(在连续出现2个table的情况下)。
|
||||
tb[0] = min(tb[0], block['bbox'][0])
|
||||
tb[1] = min(tb[1], block['bbox'][1])
|
||||
tb[2] = max(tb[2], block['bbox'][2])
|
||||
tb[3] = max(tb[3], block['bbox'][3])
|
||||
block['_table'] = True # 占位,防止其他table再次占用
|
||||
|
||||
"""如果是个table的title,但是有部分重叠,那么修正这个title,使得和table不重叠"""
|
||||
if _is_part_overlap(tb, block['bbox']) and __check_table_title_pattern(txt):
|
||||
block['bbox'] = list(block['bbox'])
|
||||
if block['bbox'][3] > U:
|
||||
block['bbox'][3] = U-1
|
||||
if block['bbox'][1] < D:
|
||||
block['bbox'][1] = D+1
|
||||
|
||||
|
||||
return table_bboxes
|
||||
|
||||
|
||||
def __get_table_caption_text(text_block):
|
||||
txt = " ".join(span['text'] for line in text_block['lines'] for span in line['spans'])
|
||||
line_cnt = len(text_block['lines'])
|
||||
txt = txt.replace("Ž . ", '')
|
||||
return txt, line_cnt
|
||||
|
||||
|
||||
def include_table_title(pymu_blocks, table_bboxes: list):
|
||||
"""
|
||||
把表格的title也包含进来,扩展到table_bbox上
|
||||
"""
|
||||
for tb in table_bboxes:
|
||||
max_find_cnt = 3 # 上上最多找3次
|
||||
temp_box = tb.copy()
|
||||
while max_find_cnt>0:
|
||||
text_block_top = find_top_nearest_text_bbox(pymu_blocks, temp_box)
|
||||
if text_block_top:
|
||||
txt, line_cnt = __get_table_caption_text(text_block_top)
|
||||
if len(txt.strip())>0:
|
||||
if not __check_table_title_pattern(txt) and max_find_cnt>0 and line_cnt<3:
|
||||
max_find_cnt = max_find_cnt -1
|
||||
temp_box[1] = text_block_top['bbox'][1]
|
||||
continue
|
||||
else:
|
||||
break
|
||||
else:
|
||||
temp_box[1] = text_block_top['bbox'][1] # 宽度不变,扩大
|
||||
max_find_cnt = max_find_cnt - 1
|
||||
else:
|
||||
break
|
||||
|
||||
max_find_cnt = 3 # 向下找
|
||||
temp_box = tb.copy()
|
||||
while max_find_cnt>0:
|
||||
text_block_bottom = find_bottom_nearest_text_bbox(pymu_blocks, temp_box)
|
||||
if text_block_bottom:
|
||||
txt, line_cnt = __get_table_caption_text(text_block_bottom)
|
||||
if len(txt.strip())>0:
|
||||
if not __check_table_title_pattern(txt) and max_find_cnt>0 and line_cnt<3:
|
||||
max_find_cnt = max_find_cnt - 1
|
||||
temp_box[3] = text_block_bottom['bbox'][3]
|
||||
continue
|
||||
else:
|
||||
break
|
||||
else:
|
||||
temp_box[3] = text_block_bottom['bbox'][3]
|
||||
max_find_cnt = max_find_cnt - 1
|
||||
else:
|
||||
break
|
||||
|
||||
if text_block_top and text_block_bottom and text_block_top.get("_table_caption", False) is False and text_block_bottom.get("_table_caption", False) is False :
|
||||
btn_text, _ = __get_table_caption_text(text_block_bottom)
|
||||
top_text, _ = __get_table_caption_text(text_block_top)
|
||||
if __check_table_title_pattern(btn_text) and __check_table_title_pattern(top_text): # 上下都有一个tbale的caption
|
||||
# 取距离最近的
|
||||
btn_text_distance = text_block_bottom['bbox'][1] - tb[3]
|
||||
top_text_distance = tb[1] - text_block_top['bbox'][3]
|
||||
text_block = text_block_bottom if btn_text_distance<top_text_distance else text_block_top
|
||||
tb[0] = min(tb[0], text_block['bbox'][0])
|
||||
tb[1] = min(tb[1], text_block['bbox'][1])
|
||||
tb[2] = max(tb[2], text_block['bbox'][2])
|
||||
tb[3] = max(tb[3], text_block['bbox'][3])
|
||||
text_block_bottom['_table_caption'] = True
|
||||
continue
|
||||
|
||||
# 如果以上条件都不满足,那么就向下找
|
||||
text_block = text_block_top
|
||||
if text_block and text_block.get("_table_caption", False) is False:
|
||||
first_text_line = " ".join(span['text'] for line in text_block['lines'] for span in line['spans'])
|
||||
if __check_table_title_pattern(first_text_line) and text_block.get("_table", False) is False:
|
||||
tb[0] = min(tb[0], text_block['bbox'][0])
|
||||
tb[1] = min(tb[1], text_block['bbox'][1])
|
||||
tb[2] = max(tb[2], text_block['bbox'][2])
|
||||
tb[3] = max(tb[3], text_block['bbox'][3])
|
||||
text_block['_table_caption'] = True
|
||||
continue
|
||||
|
||||
text_block = text_block_bottom
|
||||
if text_block and text_block.get("_table_caption", False) is False:
|
||||
first_text_line, _ = __get_table_caption_text(text_block)
|
||||
if __check_table_title_pattern(first_text_line) and text_block.get("_table", False) is False:
|
||||
tb[0] = min(tb[0], text_block['bbox'][0])
|
||||
tb[1] = min(tb[1], text_block['bbox'][1])
|
||||
tb[2] = max(tb[2], text_block['bbox'][2])
|
||||
tb[3] = max(tb[3], text_block['bbox'][3])
|
||||
text_block['_table_caption'] = True
|
||||
continue
|
||||
|
||||
"""向左、向右寻找,暂时只寻找一次"""
|
||||
left_text_block = find_left_nearest_text_bbox(pymu_blocks, tb)
|
||||
if left_text_block and left_text_block.get("_image_caption", False) is False:
|
||||
first_text_line, _ = __get_table_caption_text(left_text_block)
|
||||
if __check_table_title_pattern(first_text_line):
|
||||
tb[0] = min(tb[0], left_text_block['bbox'][0])
|
||||
tb[1] = min(tb[1], left_text_block['bbox'][1])
|
||||
tb[2] = max(tb[2], left_text_block['bbox'][2])
|
||||
tb[3] = max(tb[3], left_text_block['bbox'][3])
|
||||
left_text_block['_image_caption'] = True
|
||||
continue
|
||||
|
||||
right_text_block = find_right_nearest_text_bbox(pymu_blocks, tb)
|
||||
if right_text_block and right_text_block.get("_image_caption", False) is False:
|
||||
first_text_line, _ = __get_table_caption_text(right_text_block)
|
||||
if __check_table_title_pattern(first_text_line):
|
||||
tb[0] = min(tb[0], right_text_block['bbox'][0])
|
||||
tb[1] = min(tb[1], right_text_block['bbox'][1])
|
||||
tb[2] = max(tb[2], right_text_block['bbox'][2])
|
||||
tb[3] = max(tb[3], right_text_block['bbox'][3])
|
||||
right_text_block['_image_caption'] = True
|
||||
continue
|
||||
|
||||
return table_bboxes
|
||||
+45
@@ -0,0 +1,45 @@
|
||||
import os
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
import click
|
||||
import json
|
||||
from loguru import logger
|
||||
|
||||
from libs.commons import join_path, parse_aws_param, parse_bucket_key, read_file
|
||||
from mkcontent import mk_nlp_markdown
|
||||
from pdf2md import main
|
||||
from pdf_parse_by_model import parse_pdf_by_model
|
||||
|
||||
|
||||
|
||||
|
||||
@click.command()
|
||||
@click.option("--pdf-file-path", help="s3上pdf文件的路径")
|
||||
@click.option("--pdf-name", help="pdf name")
|
||||
def main_shell(pdf_file_path: str, pdf_name: str):
|
||||
with open('/mnt/petrelfs/share_data/ouyanglinke/OCR/OCR_validation_dataset_final_rotated_formulafix_highdpi_scihub.json', 'r') as f:
|
||||
samples = json.load(f)
|
||||
for sample in samples:
|
||||
pdf_file_path = sample['s3_path']
|
||||
pdf_bin_file_profile = "outsider"
|
||||
pdf_name = sample['pdf_name']
|
||||
pdf_model_dir = f"s3://llm-pdf-text/eval_1k/layout_res/{pdf_name}"
|
||||
pdf_model_profile = "langchao"
|
||||
|
||||
p = Path(pdf_file_path)
|
||||
pdf_file_name = p.name # pdf文件名字,含后缀
|
||||
|
||||
#pdf_model_dir = join_path(pdf_model_parent_dir, pdf_file_name)
|
||||
|
||||
main(
|
||||
pdf_file_path,
|
||||
pdf_bin_file_profile,
|
||||
pdf_model_dir,
|
||||
pdf_model_profile,
|
||||
debug_mode=True,
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main_shell()
|
||||
@@ -0,0 +1,82 @@
|
||||
# from app.common import s3
|
||||
import boto3
|
||||
from botocore.client import Config
|
||||
|
||||
from spark import s3_buckets, s3_clusters, get_cluster_name, s3_users
|
||||
import re
|
||||
import random
|
||||
from typing import Dict, Iterator, List, Tuple, Union
|
||||
|
||||
__re_s3_path = re.compile("^s3a?://([^/]+)(?:/(.*))?$")
|
||||
def get_s3_config(path: Union[str, List[str]], outside=False):
|
||||
paths = [path] if type(path) == str else path
|
||||
bucket_config = None
|
||||
for p in paths:
|
||||
bc = __get_s3_bucket_config(p)
|
||||
if bucket_config in [bc, None]:
|
||||
bucket_config = bc
|
||||
continue
|
||||
raise Exception(f"{paths} have different s3 config, cannot read together.")
|
||||
if not bucket_config:
|
||||
raise Exception("path is empty.")
|
||||
return __get_s3_config(bucket_config, outside, prefer_ip=True)
|
||||
|
||||
def __get_s3_config(
|
||||
bucket_config: tuple,
|
||||
outside: bool,
|
||||
prefer_ip=False,
|
||||
prefer_auto=False,
|
||||
):
|
||||
cluster, user = bucket_config
|
||||
cluster_config = s3_clusters[cluster]
|
||||
|
||||
if outside:
|
||||
endpoint_key = "outside"
|
||||
elif prefer_auto and "auto" in cluster_config:
|
||||
endpoint_key = "auto"
|
||||
elif cluster_config.get("cluster") == get_cluster_name():
|
||||
endpoint_key = "inside"
|
||||
else:
|
||||
endpoint_key = "outside"
|
||||
|
||||
if prefer_ip and f"{endpoint_key}_ips" in cluster_config:
|
||||
endpoint_key = f"{endpoint_key}_ips"
|
||||
|
||||
endpoints = cluster_config[endpoint_key]
|
||||
endpoint = random.choice(endpoints)
|
||||
return {"endpoint": endpoint, **s3_users[user]}
|
||||
|
||||
def split_s3_path(path: str):
|
||||
"split bucket and key from path"
|
||||
m = __re_s3_path.match(path)
|
||||
if m is None:
|
||||
return "", ""
|
||||
return m.group(1), (m.group(2) or "")
|
||||
|
||||
def __get_s3_bucket_config(path: str):
|
||||
bucket = split_s3_path(path)[0] if path else ""
|
||||
bucket_config = s3_buckets.get(bucket)
|
||||
if not bucket_config:
|
||||
bucket_config = s3_buckets.get("[default]")
|
||||
assert bucket_config is not None
|
||||
return bucket_config
|
||||
|
||||
def get_s3_client(path: Union[str, List[str]], outside=False):
|
||||
s3_config = get_s3_config(path, outside)
|
||||
try:
|
||||
return boto3.client(
|
||||
"s3",
|
||||
aws_access_key_id=s3_config["ak"],
|
||||
aws_secret_access_key=s3_config["sk"],
|
||||
endpoint_url=s3_config["endpoint"],
|
||||
config=Config(s3={"addressing_style": "path"}, retries={"max_attempts": 8, "mode": "standard"}),
|
||||
)
|
||||
except:
|
||||
# older boto3 do not support retries.mode param.
|
||||
return boto3.client(
|
||||
"s3",
|
||||
aws_access_key_id=s3_config["ak"],
|
||||
aws_secret_access_key=s3_config["sk"],
|
||||
endpoint_url=s3_config["endpoint"],
|
||||
config=Config(s3={"addressing_style": "path"}, retries={"max_attempts": 8}),
|
||||
)
|
||||
@@ -0,0 +1,17 @@
|
||||
scihub/scihub_00500000/libgen.scimag00527000-00527999.zip_10.1002/app.25178
|
||||
scihub/scihub_07400000/libgen.scimag07481000-07481999.zip_10.1007/s003960050343
|
||||
scihub/scihub_11400000/libgen.scimag11451000-11451999.zip_10.1017/s0009838811000231
|
||||
scihub/scihub_24400000/libgen.scimag24401000-24401999.zip_10.1016/j.toxicon.2014.02.018
|
||||
scihub/scihub_27400000/libgen.scimag27441000-27441999.zip_10.2307/30122482
|
||||
scihub/scihub_28400000/libgen.scimag28413000-28413999.zip_10.2307/1316224
|
||||
scihub/scihub_31200000/libgen.scimag31207000-31207999.zip_10.1080/03639040600920622
|
||||
scihub/scihub_31800000/libgen.scimag31824000-31824999.zip_10.1109/med.2012.6265668
|
||||
scihub/scihub_32500000/libgen.scimag32539000-32539999.zip_10.1080/09540121003721000
|
||||
scihub/scihub_42500000/libgen.scimag42522000-42522999.zip_10.1016/S1365-6937(15)30162-3
|
||||
scihub/scihub_45900000/libgen.scimag45914000-45914999.zip_10.1055/s-0030-1256333
|
||||
scihub/scihub_50900000/libgen.scimag50902000-50902999.zip_10.1007/s12274-016-1035-8
|
||||
scihub/scihub_63900000/libgen.scimag63921000-63921999.zip_10.1063/1.4938050
|
||||
scihub/scihub_65800000/libgen.scimag65832000-65832999.zip_10.1016/s0166-4115(08)62165-2
|
||||
scihub/scihub_67300000/libgen.scimag67369000-67369999.zip_10.1096/fj.201700997R
|
||||
scihub/scihub_67900000/libgen.scimag67967000-67967999.zip_10.1038/s41598-018-21867-z
|
||||
scihub/scihub_77400000/libgen.scimag77447000-77447999.zip_10.1016/j.jid.2019.06.094
|
||||
@@ -0,0 +1,112 @@
|
||||
{
|
||||
"pageID_imageBboxs": [
|
||||
[
|
||||
[
|
||||
37.5,
|
||||
651.3200073242188,
|
||||
557.5,
|
||||
790.8300170898438
|
||||
]
|
||||
],
|
||||
[
|
||||
[
|
||||
69,
|
||||
771,
|
||||
119,
|
||||
789
|
||||
]
|
||||
],
|
||||
[],
|
||||
[
|
||||
[
|
||||
152.16000366210938,
|
||||
188.3996124267578,
|
||||
458.8800048828125,
|
||||
330.4796142578125
|
||||
],
|
||||
[
|
||||
200.57258064516128,
|
||||
499.0322580645161,
|
||||
339.67863247863244,
|
||||
627.5418803418803
|
||||
]
|
||||
],
|
||||
[
|
||||
[
|
||||
88.29032258064515,
|
||||
246.63709677419354,
|
||||
288.822792022792,
|
||||
608.8307692307692
|
||||
]
|
||||
],
|
||||
[
|
||||
[
|
||||
168.42338709677418,
|
||||
118.04032258064515,
|
||||
439.9509971509971,
|
||||
246.60284900284898
|
||||
],
|
||||
[
|
||||
155.94758064516128,
|
||||
341.1653225806451,
|
||||
452.4250712250712,
|
||||
434.67350427350425
|
||||
]
|
||||
],
|
||||
[],
|
||||
[]
|
||||
],
|
||||
"pageID_tableBboxs": [
|
||||
[],
|
||||
[],
|
||||
[],
|
||||
[],
|
||||
[],
|
||||
[
|
||||
[
|
||||
71.9758064516129,
|
||||
534.0604838709677,
|
||||
520.5527065527065,
|
||||
717.7390313390313
|
||||
]
|
||||
],
|
||||
[
|
||||
[
|
||||
71.01612903225806,
|
||||
296.5403225806451,
|
||||
525.3504273504273,
|
||||
392.45356125356125
|
||||
],
|
||||
[
|
||||
79.17338709677419,
|
||||
503.8306451612903,
|
||||
517.674074074074,
|
||||
736.4501424501424
|
||||
]
|
||||
],
|
||||
[]
|
||||
],
|
||||
"pageID_equationBboxs": [
|
||||
[],
|
||||
[],
|
||||
[
|
||||
[
|
||||
247.59677419354836,
|
||||
511.9879032258064,
|
||||
524.8706552706552,
|
||||
525.8301994301994
|
||||
],
|
||||
[
|
||||
222.16532258064515,
|
||||
588.7620967741935,
|
||||
524.3908831908832,
|
||||
608.8307692307692
|
||||
]
|
||||
],
|
||||
[],
|
||||
[],
|
||||
[],
|
||||
[],
|
||||
[]
|
||||
]
|
||||
}
|
||||
Binary file not shown.
File diff suppressed because it is too large
Load Diff
Binary file not shown.
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,271 @@
|
||||
{
|
||||
"page_0":{
|
||||
"para_blocks": [
|
||||
{
|
||||
"block_id": 0,
|
||||
"bbox": [39.0, 34.719993591308594, 347.1359558105469, 51.2079963684082],
|
||||
"text": "IOP Conference Series: Earth and Environmental Science",
|
||||
"dir": [1.0, 0.0],
|
||||
"X0": 39.0,
|
||||
"X1": 347.1359558105469,
|
||||
"avg_char_width": 6.4194990793863935,
|
||||
"avg_char_height": 16.48800277709961,
|
||||
"block_font_type": "Helvetica",
|
||||
"block_font_size": 12.0,
|
||||
"is_segmented": 1,
|
||||
"paras": [
|
||||
{
|
||||
"para_id": 0,
|
||||
"bbox": [39.0, 34.719993591308594, 347.1359558105469, 51.2079963684082],
|
||||
"text": "IOP Conference Series: Earth and Environmental Science",
|
||||
"is_matched": 1,
|
||||
"is_title": 0,
|
||||
"font_type": "Helvetica",
|
||||
"font_size": 12.0,
|
||||
"font_color": 0,
|
||||
"neighbor_paras": [null, null]
|
||||
}
|
||||
],
|
||||
"bboxes_para": [[39.0, 34.719993591308594, 347.1359558105469, 51.2079963684082]]
|
||||
},
|
||||
{
|
||||
"block_id": 1,
|
||||
"bbox": [39.0, 111.38001251220703, 143.67001342773438, 123.77301025390625],
|
||||
"text": "PAPER • OPEN ACCESS",
|
||||
"dir": [1.0, 0.0],
|
||||
"X0": 39.0,
|
||||
"X1": 143.67001342773438,
|
||||
"avg_char_width": 6.541875839233398,
|
||||
"avg_char_height": 12.392997741699219,
|
||||
"block_font_type": "Helvetica-Bold",
|
||||
"block_font_size": 9.0,
|
||||
"is_segmented": 1,
|
||||
"paras": [
|
||||
{
|
||||
"para_id": 0,
|
||||
"bbox": [39.0, 111.38001251220703, 143.67001342773438, 123.77301025390625],
|
||||
"text": "PAPER • OPEN ACCESS",
|
||||
"is_matched": 1,
|
||||
"is_title": 0,
|
||||
"font_type": "Helvetica-Bold",
|
||||
"font_size": 9.0,
|
||||
"font_color": 0,
|
||||
"neighbor_paras": [null, null]
|
||||
},
|
||||
{
|
||||
"para_id": 1,
|
||||
"bbox": [39.0, 111.38001251220703, 143.67001342773438, 123.77301025390625],
|
||||
"text": "PAPER • OPEN ACCESS",
|
||||
"is_matched": 1,
|
||||
"is_title": 0,
|
||||
"font_type": "Helvetica-Bold",
|
||||
"font_size": 9.0,
|
||||
"font_color": 0,
|
||||
"neighbor_paras": [null, null]
|
||||
}
|
||||
],
|
||||
"bboxes_para": [[39.0, 111.38001251220703, 143.67001342773438, 123.77301025390625]]
|
||||
}
|
||||
],
|
||||
|
||||
"line_blocks":[
|
||||
{
|
||||
"number": 0,
|
||||
"type": 0,
|
||||
"bbox": [
|
||||
428.93170166015625,
|
||||
744.921142578125,
|
||||
541.5675048828125,
|
||||
757.8131713867188
|
||||
],
|
||||
"lines": [
|
||||
{
|
||||
"spans": [
|
||||
{
|
||||
"size": 11.0,
|
||||
"flags": 20,
|
||||
"font": "UniversNextPro-BoldCond",
|
||||
"color": 0,
|
||||
"ascender": 0.9490000009536743,
|
||||
"descender": -0.22300000488758087,
|
||||
"text": "3",
|
||||
"origin": [
|
||||
536.37548828125,
|
||||
755.3601684570312
|
||||
],
|
||||
"bbox": [
|
||||
536.37548828125,
|
||||
744.921142578125,
|
||||
541.5675048828125,
|
||||
757.8131713867188
|
||||
]
|
||||
}
|
||||
],
|
||||
"wmode": 0,
|
||||
"dir": [
|
||||
1.0,
|
||||
0.0
|
||||
],
|
||||
"bbox": [
|
||||
536.37548828125,
|
||||
744.921142578125,
|
||||
541.5675048828125,
|
||||
757.8131713867188
|
||||
]
|
||||
},
|
||||
{
|
||||
"spans": [
|
||||
{
|
||||
"size": 8.0,
|
||||
"flags": 20,
|
||||
"font": "UniversNextPro-BoldCond",
|
||||
"color": 0,
|
||||
"ascender": 0.9490000009536743,
|
||||
"descender": -0.22300000488758087,
|
||||
"text": "Spektrum ",
|
||||
"origin": [
|
||||
428.93170166015625,
|
||||
755.3601684570312
|
||||
],
|
||||
"bbox": [
|
||||
428.93170166015625,
|
||||
747.7681884765625,
|
||||
458.7516174316406,
|
||||
757.1441650390625
|
||||
]
|
||||
},
|
||||
{
|
||||
"size": 8.0,
|
||||
"flags": 4,
|
||||
"font": "UniversNextPro-Cond",
|
||||
"color": 0,
|
||||
"ascender": 0.9359999895095825,
|
||||
"descender": -0.21400000154972076,
|
||||
"text": "der Wissenschaft ",
|
||||
"origin": [
|
||||
458.431884765625,
|
||||
755.3601684570312
|
||||
],
|
||||
"bbox": [
|
||||
458.431884765625,
|
||||
747.8721923828125,
|
||||
508.0399169921875,
|
||||
757.0721435546875
|
||||
]
|
||||
},
|
||||
{
|
||||
"size": 8.0,
|
||||
"flags": 4,
|
||||
"font": "UniversNextPro-Regular",
|
||||
"color": 0,
|
||||
"ascender": 0.9290000200271606,
|
||||
"descender": -0.22200000286102295,
|
||||
"text": "7.21",
|
||||
"origin": [
|
||||
510.2349853515625,
|
||||
755.3601684570312
|
||||
],
|
||||
"bbox": [
|
||||
510.2349853515625,
|
||||
747.9281616210938,
|
||||
524.5621948242188,
|
||||
757.1361694335938
|
||||
]
|
||||
}
|
||||
],
|
||||
"wmode": 0,
|
||||
"dir": [
|
||||
1.0,
|
||||
0.0
|
||||
],
|
||||
"bbox": [
|
||||
428.93170166015625,
|
||||
747.7681884765625,
|
||||
524.5621948242188,
|
||||
757.1441650390625
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
|
||||
"images":[
|
||||
{
|
||||
"image_bbox":[0,0,1,1],
|
||||
"image_path":"path/to/image.jpg"
|
||||
},
|
||||
{
|
||||
"image_bbox":[1,2,3,4],
|
||||
"image_path":"path/to/image.jpg"
|
||||
}
|
||||
],
|
||||
|
||||
"tables":[
|
||||
{
|
||||
"table_bbox":[0,0,1,1],
|
||||
"image_path":"path/to/image.jpg"
|
||||
},
|
||||
{
|
||||
"table_bbox":[1,2,3,4],
|
||||
"image_path":"path/to/image.jpg"
|
||||
}
|
||||
],
|
||||
|
||||
"interline_equations":[
|
||||
{
|
||||
"equation_bbox":[0,0,1,1],
|
||||
"image_path":"path/to/equation.jpg"
|
||||
},
|
||||
{
|
||||
"equation_bbox":[1,2,3,4],
|
||||
"image_path":"path/to/equation.jpg"
|
||||
}
|
||||
],
|
||||
|
||||
"inline_equations":[
|
||||
{
|
||||
"equation_bbox":[0,0,1,1],
|
||||
"image_path":"path/to/equation.jpg"
|
||||
},
|
||||
{
|
||||
"equation_bbox":[1,2,3,4],
|
||||
"image_path":"path/to/equation.jpg"
|
||||
}
|
||||
],
|
||||
|
||||
"layout_bboxes":[
|
||||
{
|
||||
"layout_bbox": [0,0, 1,1],
|
||||
"layout_label":"V|H|B"
|
||||
},
|
||||
{
|
||||
"layout_bbox": [1,2,3,4],
|
||||
"layout_label":"V|H|B"
|
||||
}
|
||||
],
|
||||
"pymu_raw_blocks":[],
|
||||
|
||||
"global_statistic":{
|
||||
|
||||
},
|
||||
"droped_text_block":[
|
||||
|
||||
|
||||
],
|
||||
"droped_image_block":[
|
||||
|
||||
],
|
||||
"droped_table_block":[
|
||||
|
||||
],
|
||||
"image_backup":[
|
||||
|
||||
],
|
||||
"table_backup":[
|
||||
|
||||
]
|
||||
},
|
||||
"page_1":{
|
||||
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,617 @@
|
||||
[
|
||||
{
|
||||
"number": 0,
|
||||
"type": 0,
|
||||
"bbox": [
|
||||
133.7009735107422,
|
||||
71.23444366455078,
|
||||
241.90069580078125,
|
||||
105.01055908203125
|
||||
],
|
||||
"lines": [
|
||||
{
|
||||
"spans": [
|
||||
{
|
||||
"size": 25.81100082397461,
|
||||
"flags": 0,
|
||||
"font": "ËÎÌå",
|
||||
"color": 0,
|
||||
"ascender": 1.04296875,
|
||||
"descender": -0.265625,
|
||||
"text": "调虎离山",
|
||||
"origin": [
|
||||
133.7009735107422,
|
||||
98.15451049804688
|
||||
],
|
||||
"bbox": [
|
||||
133.7009735107422,
|
||||
71.23444366455078,
|
||||
241.90069580078125,
|
||||
105.01055908203125
|
||||
]
|
||||
}
|
||||
],
|
||||
"wmode": 0,
|
||||
"dir": [
|
||||
1.0,
|
||||
0.0
|
||||
],
|
||||
"bbox": [
|
||||
133.7009735107422,
|
||||
71.23444366455078,
|
||||
241.90069580078125,
|
||||
105.01055908203125
|
||||
]
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"number": 1,
|
||||
"type": 0,
|
||||
"bbox": [
|
||||
45.990631103515625,
|
||||
298.78173828125,
|
||||
93.81614685058594,
|
||||
314.27288818359375
|
||||
],
|
||||
"lines": [
|
||||
{
|
||||
"spans": [
|
||||
{
|
||||
"size": 11.838000297546387,
|
||||
"flags": 0,
|
||||
"font": "ËÎÌå",
|
||||
"color": 0,
|
||||
"ascender": 1.04296875,
|
||||
"descender": -0.265625,
|
||||
"text": "【译文】",
|
||||
"origin": [
|
||||
45.990631103515625,
|
||||
311.12841796875
|
||||
],
|
||||
"bbox": [
|
||||
45.990631103515625,
|
||||
298.78173828125,
|
||||
93.81614685058594,
|
||||
314.27288818359375
|
||||
]
|
||||
}
|
||||
],
|
||||
"wmode": 0,
|
||||
"dir": [
|
||||
1.0,
|
||||
0.0
|
||||
],
|
||||
"bbox": [
|
||||
45.990631103515625,
|
||||
298.78173828125,
|
||||
93.81614685058594,
|
||||
314.27288818359375
|
||||
]
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"number": 2,
|
||||
"type": 0,
|
||||
"bbox": [
|
||||
76.05915069580078,
|
||||
323.13250732421875,
|
||||
324.24285888671875,
|
||||
338.6236572265625
|
||||
],
|
||||
"lines": [
|
||||
{
|
||||
"spans": [
|
||||
{
|
||||
"size": 11.838000297546387,
|
||||
"flags": 0,
|
||||
"font": "ËÎÌå",
|
||||
"color": 0,
|
||||
"ascender": 1.04296875,
|
||||
"descender": -0.265625,
|
||||
"text": "等待客观条件对敌方不利时再去围困它,用人为",
|
||||
"origin": [
|
||||
76.05915069580078,
|
||||
335.47918701171875
|
||||
],
|
||||
"bbox": [
|
||||
76.05915069580078,
|
||||
323.13250732421875,
|
||||
324.24285888671875,
|
||||
338.6236572265625
|
||||
]
|
||||
}
|
||||
],
|
||||
"wmode": 0,
|
||||
"dir": [
|
||||
1.0,
|
||||
0.0
|
||||
],
|
||||
"bbox": [
|
||||
76.05915069580078,
|
||||
323.13250732421875,
|
||||
324.24285888671875,
|
||||
338.6236572265625
|
||||
]
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"number": 3,
|
||||
"type": 0,
|
||||
"bbox": [
|
||||
51.968841552734375,
|
||||
347.19915771484375,
|
||||
324.2073669433594,
|
||||
362.6903076171875
|
||||
],
|
||||
"lines": [
|
||||
{
|
||||
"spans": [
|
||||
{
|
||||
"size": 11.838000297546387,
|
||||
"flags": 0,
|
||||
"font": "ËÎÌå",
|
||||
"color": 0,
|
||||
"ascender": 1.04296875,
|
||||
"descender": -0.265625,
|
||||
"text": "的因素去诱惑调动它。向前进攻有危险时,就想办法",
|
||||
"origin": [
|
||||
51.968841552734375,
|
||||
359.54583740234375
|
||||
],
|
||||
"bbox": [
|
||||
51.968841552734375,
|
||||
347.19915771484375,
|
||||
324.2073669433594,
|
||||
362.6903076171875
|
||||
]
|
||||
}
|
||||
],
|
||||
"wmode": 0,
|
||||
"dir": [
|
||||
1.0,
|
||||
0.0
|
||||
],
|
||||
"bbox": [
|
||||
51.968841552734375,
|
||||
347.19915771484375,
|
||||
324.2073669433594,
|
||||
362.6903076171875
|
||||
]
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"number": 4,
|
||||
"type": 0,
|
||||
"bbox": [
|
||||
51.933349609375,
|
||||
371.36053466796875,
|
||||
159.9906005859375,
|
||||
386.8516540527344
|
||||
],
|
||||
"lines": [
|
||||
{
|
||||
"spans": [
|
||||
{
|
||||
"size": 11.838000297546387,
|
||||
"flags": 0,
|
||||
"font": "ËÎÌå",
|
||||
"color": 0,
|
||||
"ascender": 1.04296875,
|
||||
"descender": -0.265625,
|
||||
"text": "让敌人返过来攻我。",
|
||||
"origin": [
|
||||
51.933349609375,
|
||||
383.7071838378906
|
||||
],
|
||||
"bbox": [
|
||||
51.933349609375,
|
||||
371.36053466796875,
|
||||
159.9906005859375,
|
||||
386.8516540527344
|
||||
]
|
||||
}
|
||||
],
|
||||
"wmode": 0,
|
||||
"dir": [
|
||||
1.0,
|
||||
0.0
|
||||
],
|
||||
"bbox": [
|
||||
51.933349609375,
|
||||
371.36053466796875,
|
||||
159.9906005859375,
|
||||
386.8516540527344
|
||||
]
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"number": 5,
|
||||
"type": 0,
|
||||
"bbox": [
|
||||
45.884117126464844,
|
||||
402.60101318359375,
|
||||
93.70964050292969,
|
||||
418.0921325683594
|
||||
],
|
||||
"lines": [
|
||||
{
|
||||
"spans": [
|
||||
{
|
||||
"size": 11.838000297546387,
|
||||
"flags": 0,
|
||||
"font": "ËÎÌå",
|
||||
"color": 0,
|
||||
"ascender": 1.04296875,
|
||||
"descender": -0.265625,
|
||||
"text": "【溯源】",
|
||||
"origin": [
|
||||
45.884117126464844,
|
||||
414.9476623535156
|
||||
],
|
||||
"bbox": [
|
||||
45.884117126464844,
|
||||
402.60101318359375,
|
||||
93.70964050292969,
|
||||
418.0921325683594
|
||||
]
|
||||
}
|
||||
],
|
||||
"wmode": 0,
|
||||
"dir": [
|
||||
1.0,
|
||||
0.0
|
||||
],
|
||||
"bbox": [
|
||||
45.884117126464844,
|
||||
402.60101318359375,
|
||||
93.70964050292969,
|
||||
418.0921325683594
|
||||
]
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"number": 6,
|
||||
"type": 0,
|
||||
"bbox": [
|
||||
75.95264434814453,
|
||||
426.9517822265625,
|
||||
324.14825439453125,
|
||||
442.4429016113281
|
||||
],
|
||||
"lines": [
|
||||
{
|
||||
"spans": [
|
||||
{
|
||||
"size": 11.838000297546387,
|
||||
"flags": 0,
|
||||
"font": "ËÎÌå",
|
||||
"color": 0,
|
||||
"ascender": 1.04296875,
|
||||
"descender": -0.265625,
|
||||
"text": "调虎离山,此计用在军事上,是一种调动敌人的",
|
||||
"origin": [
|
||||
75.95264434814453,
|
||||
439.2984313964844
|
||||
],
|
||||
"bbox": [
|
||||
75.95264434814453,
|
||||
426.9517822265625,
|
||||
324.14825439453125,
|
||||
442.4429016113281
|
||||
]
|
||||
}
|
||||
],
|
||||
"wmode": 0,
|
||||
"dir": [
|
||||
1.0,
|
||||
0.0
|
||||
],
|
||||
"bbox": [
|
||||
75.95264434814453,
|
||||
426.9517822265625,
|
||||
324.14825439453125,
|
||||
442.4429016113281
|
||||
]
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"number": 7,
|
||||
"type": 0,
|
||||
"bbox": [
|
||||
169.9068603515625,
|
||||
447.10321044921875,
|
||||
205.61537170410156,
|
||||
459.161376953125
|
||||
],
|
||||
"lines": [
|
||||
{
|
||||
"spans": [
|
||||
{
|
||||
"size": 8.829999923706055,
|
||||
"flags": 4,
|
||||
"font": "DY2+ZJSBrL-2",
|
||||
"color": 0,
|
||||
"ascender": 0.800000011920929,
|
||||
"descender": -0.20000000298023224,
|
||||
"text": "!",
|
||||
"origin": [
|
||||
185.62425231933594,
|
||||
456.81591796875
|
||||
],
|
||||
"bbox": [
|
||||
185.62425231933594,
|
||||
449.7519226074219,
|
||||
190.03924560546875,
|
||||
458.5819091796875
|
||||
]
|
||||
},
|
||||
{
|
||||
"size": 8.829999923706055,
|
||||
"flags": 0,
|
||||
"font": "ËÎÌå",
|
||||
"color": 0,
|
||||
"ascender": 1.04296875,
|
||||
"descender": -0.265625,
|
||||
"text": " —",
|
||||
"origin": [
|
||||
190.03924560546875,
|
||||
456.81591796875
|
||||
],
|
||||
"bbox": [
|
||||
190.03924560546875,
|
||||
447.10321044921875,
|
||||
205.61537170410156,
|
||||
459.161376953125
|
||||
]
|
||||
}
|
||||
],
|
||||
"wmode": 0,
|
||||
"dir": [
|
||||
1.0,
|
||||
0.0
|
||||
],
|
||||
"bbox": [
|
||||
185.62425231933594,
|
||||
447.10321044921875,
|
||||
205.61537170410156,
|
||||
459.161376953125
|
||||
]
|
||||
},
|
||||
{
|
||||
"spans": [
|
||||
{
|
||||
"size": 8.829999923706055,
|
||||
"flags": 0,
|
||||
"font": "ËÎÌå",
|
||||
"color": 0,
|
||||
"ascender": 1.04296875,
|
||||
"descender": -0.265625,
|
||||
"text": "—",
|
||||
"origin": [
|
||||
169.9068603515625,
|
||||
456.3126220703125
|
||||
],
|
||||
"bbox": [
|
||||
169.9068603515625,
|
||||
447.10321044921875,
|
||||
178.7368621826172,
|
||||
458.6580810546875
|
||||
]
|
||||
}
|
||||
],
|
||||
"wmode": 0,
|
||||
"dir": [
|
||||
1.0,
|
||||
0.0
|
||||
],
|
||||
"bbox": [
|
||||
169.9068603515625,
|
||||
447.10321044921875,
|
||||
178.7368621826172,
|
||||
458.6580810546875
|
||||
]
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"number": 8,
|
||||
"type": 0,
|
||||
"bbox": [
|
||||
106.35260009765625,
|
||||
11.284072875976562,
|
||||
238.4646453857422,
|
||||
26.775205612182617
|
||||
],
|
||||
"lines": [
|
||||
{
|
||||
"spans": [
|
||||
{
|
||||
"size": 11.838000297546387,
|
||||
"flags": 0,
|
||||
"font": "ËÎÌå",
|
||||
"color": 16777215,
|
||||
"ascender": 1.04296875,
|
||||
"descender": -0.265625,
|
||||
"text": "图说案例本三十六计全书",
|
||||
"origin": [
|
||||
106.35260009765625,
|
||||
23.6307373046875
|
||||
],
|
||||
"bbox": [
|
||||
106.35260009765625,
|
||||
11.284072875976562,
|
||||
238.4646453857422,
|
||||
26.775205612182617
|
||||
]
|
||||
}
|
||||
],
|
||||
"wmode": 0,
|
||||
"dir": [
|
||||
1.0,
|
||||
0.0
|
||||
],
|
||||
"bbox": [
|
||||
106.35260009765625,
|
||||
11.284072875976562,
|
||||
238.4646453857422,
|
||||
26.775205612182617
|
||||
]
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"number": 9,
|
||||
"type": 0,
|
||||
"bbox": [
|
||||
338.5431213378906,
|
||||
231.63661193847656,
|
||||
350.3811340332031,
|
||||
283.2099609375
|
||||
],
|
||||
"lines": [
|
||||
{
|
||||
"spans": [
|
||||
{
|
||||
"size": 11.838000297546387,
|
||||
"flags": 0,
|
||||
"font": "ËÎÌå",
|
||||
"color": 16777215,
|
||||
"ascender": 1.04296875,
|
||||
"descender": -0.265625,
|
||||
"text": "第",
|
||||
"origin": [
|
||||
338.5431213378906,
|
||||
243.9832763671875
|
||||
],
|
||||
"bbox": [
|
||||
338.5431213378906,
|
||||
231.63661193847656,
|
||||
350.3811340332031,
|
||||
247.12774658203125
|
||||
]
|
||||
}
|
||||
],
|
||||
"wmode": 0,
|
||||
"dir": [
|
||||
1.0,
|
||||
0.0
|
||||
],
|
||||
"bbox": [
|
||||
338.5431213378906,
|
||||
231.63661193847656,
|
||||
350.3811340332031,
|
||||
247.12774658203125
|
||||
]
|
||||
},
|
||||
{
|
||||
"spans": [
|
||||
{
|
||||
"size": 11.838000297546387,
|
||||
"flags": 0,
|
||||
"font": "ËÎÌå",
|
||||
"color": 16777215,
|
||||
"ascender": 1.04296875,
|
||||
"descender": -0.265625,
|
||||
"text": "十",
|
||||
"origin": [
|
||||
338.5431213378906,
|
||||
256.01068115234375
|
||||
],
|
||||
"bbox": [
|
||||
338.5431213378906,
|
||||
243.6640167236328,
|
||||
350.3811340332031,
|
||||
259.1551513671875
|
||||
]
|
||||
}
|
||||
],
|
||||
"wmode": 0,
|
||||
"dir": [
|
||||
1.0,
|
||||
0.0
|
||||
],
|
||||
"bbox": [
|
||||
338.5431213378906,
|
||||
243.6640167236328,
|
||||
350.3811340332031,
|
||||
259.1551513671875
|
||||
]
|
||||
},
|
||||
{
|
||||
"spans": [
|
||||
{
|
||||
"size": 11.838000297546387,
|
||||
"flags": 0,
|
||||
"font": "ËÎÌå",
|
||||
"color": 16777215,
|
||||
"ascender": 1.04296875,
|
||||
"descender": -0.265625,
|
||||
"text": "五",
|
||||
"origin": [
|
||||
338.5431213378906,
|
||||
268.0380859375
|
||||
],
|
||||
"bbox": [
|
||||
338.5431213378906,
|
||||
255.69142150878906,
|
||||
350.3811340332031,
|
||||
271.18255615234375
|
||||
]
|
||||
}
|
||||
],
|
||||
"wmode": 0,
|
||||
"dir": [
|
||||
1.0,
|
||||
0.0
|
||||
],
|
||||
"bbox": [
|
||||
338.5431213378906,
|
||||
255.69142150878906,
|
||||
350.3811340332031,
|
||||
271.18255615234375
|
||||
]
|
||||
},
|
||||
{
|
||||
"spans": [
|
||||
{
|
||||
"size": 11.838000297546387,
|
||||
"flags": 0,
|
||||
"font": "ËÎÌå",
|
||||
"color": 16777215,
|
||||
"ascender": 1.04296875,
|
||||
"descender": -0.265625,
|
||||
"text": "计",
|
||||
"origin": [
|
||||
338.5431213378906,
|
||||
280.06549072265625
|
||||
],
|
||||
"bbox": [
|
||||
338.5431213378906,
|
||||
267.71881103515625,
|
||||
350.3811340332031,
|
||||
283.2099609375
|
||||
]
|
||||
}
|
||||
],
|
||||
"wmode": 0,
|
||||
"dir": [
|
||||
1.0,
|
||||
0.0
|
||||
],
|
||||
"bbox": [
|
||||
338.5431213378906,
|
||||
267.71881103515625,
|
||||
350.3811340332031,
|
||||
283.2099609375
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user