detect_footnote.py 8.0 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173
  1. import os
  2. from collections import Counter
  3. import re # 正则
  4. from libs.commons import fitz # pyMuPDF库
  5. import json # json
  6. def parse_footnotes_by_model(page_ID: int, page: fitz.Page, json_from_DocXchain_obj: dict, md_bookname_save_path, debug_mode=False):
  7. """
  8. :param page_ID: int类型,当前page在当前pdf文档中是第page_D页。
  9. :param page :fitz读取的当前页的内容
  10. :param res_dir_path: str类型,是每一个pdf文档,在当前.py文件的目录下生成一个与pdf文档同名的文件夹,res_dir_path就是文件夹的dir
  11. :param json_from_DocXchain_obj: dict类型,把pdf文档送入DocXChain模型中后,提取bbox,结果保存到pdf文档同名文件夹下的 page_ID.json文件中了。json_from_DocXchain_obj就是打开后的dict
  12. """
  13. DPI = 72 # use this resolution
  14. pix = page.get_pixmap(dpi=DPI)
  15. pageL = 0
  16. pageR = int(pix.w)
  17. pageU = 0
  18. pageD = int(pix.h)
  19. #--------- 通过json_from_DocXchain来获取 footnote ---------#
  20. footnote_bbox_from_DocXChain = []
  21. xf_json = json_from_DocXchain_obj
  22. width_from_json = xf_json['page_info']['width']
  23. height_from_json = xf_json['page_info']['height']
  24. LR_scaleRatio = width_from_json / (pageR - pageL)
  25. UD_scaleRatio = height_from_json / (pageD - pageU)
  26. # {0: 'title', # 标题
  27. # 1: 'figure', # 图片
  28. # 2: 'plain text', # 文本
  29. # 3: 'header', # 页眉
  30. # 4: 'page number', # 页码
  31. # 5: 'footnote', # 脚注
  32. # 6: 'footer', # 页脚
  33. # 7: 'table', # 表格
  34. # 8: 'table caption', # 表格描述
  35. # 9: 'figure caption', # 图片描述
  36. # 10: 'equation', # 公式
  37. # 11: 'full column', # 单栏
  38. # 12: 'sub column', # 多栏
  39. # 13: 'embedding', # 嵌入公式
  40. # 14: 'isolated'} # 单行公式
  41. for xf in xf_json['layout_dets']:
  42. L = xf['poly'][0] / LR_scaleRatio
  43. U = xf['poly'][1] / UD_scaleRatio
  44. R = xf['poly'][2] / LR_scaleRatio
  45. D = xf['poly'][5] / UD_scaleRatio
  46. # L += pageL # 有的页面,artBox偏移了。不在(0,0)
  47. # R += pageL
  48. # U += pageU
  49. # D += pageU
  50. L, R = min(L, R), max(L, R)
  51. U, D = min(U, D), max(U, D)
  52. # if xf['category_id'] == 5 and xf['score'] >= 0.3:
  53. if xf['category_id'] == 5 and xf['score'] >= 0.43: # 新的footnote阈值
  54. footnote_bbox_from_DocXChain.append((L, U, R, D))
  55. footnote_final_names = []
  56. footnote_final_bboxs = []
  57. footnote_ID = 0
  58. for L, U, R, D in footnote_bbox_from_DocXChain:
  59. if debug_mode:
  60. # cur_footnote = page.get_pixmap(clip=(L,U,R,D))
  61. new_footnote_name = "footnote_{}_{}.png".format(page_ID, footnote_ID) # 脚注name
  62. # cur_footnote.save(md_bookname_save_path + '/' + new_footnote_name) # 把脚注存储在新建的文件夹,并命名
  63. footnote_final_names.append(new_footnote_name) # 把脚注的名字存在list中
  64. footnote_final_bboxs.append((L, U, R, D))
  65. footnote_ID += 1
  66. footnote_final_bboxs.sort(key = lambda LURD: (LURD[1], LURD[0]))
  67. curPage_all_footnote_bboxs = footnote_final_bboxs
  68. return curPage_all_footnote_bboxs
  69. def need_remove(block):
  70. if 'lines' in block and len(block['lines']) > 0:
  71. # block中只有一行,且该行文本全是大写字母,或字体为粗体bold关键词,SB关键词,把这个block捞回来
  72. if len(block['lines']) == 1:
  73. if 'spans' in block['lines'][0] and len(block['lines'][0]['spans']) == 1:
  74. font_keywords = ['SB', 'bold', 'Bold']
  75. if block['lines'][0]['spans'][0]['text'].isupper() or any(keyword in block['lines'][0]['spans'][0]['font'] for keyword in font_keywords):
  76. return True
  77. for line in block['lines']:
  78. if 'spans' in line and len(line['spans']) > 0:
  79. for span in line['spans']:
  80. # 检测"keyword"是否在span中,忽略大小写
  81. if "keyword" in span['text'].lower():
  82. return True
  83. return False
  84. def parse_footnotes_by_rule(remain_text_blocks, page_height, page_id, main_text_font):
  85. """
  86. 根据给定的文本块、页高和页码,解析出符合规则的脚注文本块,并返回其边界框。
  87. Args:
  88. remain_text_blocks (list): 包含所有待处理的文本块的列表。
  89. page_height (float): 页面的高度。
  90. page_id (int): 页面的ID。
  91. Returns:
  92. list: 符合规则的脚注文本块的边界框列表。
  93. """
  94. if page_id > 20:
  95. return []
  96. else:
  97. # 存储每一行的文本块大小的列表
  98. line_sizes = []
  99. # 存储每个文本块的平均行大小
  100. block_sizes = []
  101. # 存储每一行的字体信息
  102. # font_names = []
  103. font_names = Counter()
  104. if len(remain_text_blocks) > 0:
  105. for block in remain_text_blocks:
  106. block_line_sizes = []
  107. # block_fonts = []
  108. block_fonts = Counter()
  109. for line in block['lines']:
  110. # 提取每个span的size属性,并计算行大小
  111. span_sizes = [span['size'] for span in line['spans'] if 'size' in span]
  112. if span_sizes:
  113. line_size = sum(span_sizes) / len(span_sizes)
  114. line_sizes.append(line_size)
  115. block_line_sizes.append(line_size)
  116. span_font = [(span['font'], len(span['text'])) for span in line['spans'] if 'font' in span and len(span['text']) > 0]
  117. if span_font:
  118. # # todo main_text_font应该用基于字数最多的字体而不是span级别的统计
  119. # font_names.append(font_name for font_name in span_font)
  120. # block_fonts.append(font_name for font_name in span_font)
  121. for font, count in span_font:
  122. # font_names.extend([font] * count)
  123. # block_fonts.extend([font] * count)
  124. font_names[font] += count
  125. block_fonts[font] += count
  126. if block_line_sizes:
  127. # 计算文本块的平均行大小
  128. block_size = sum(block_line_sizes) / len(block_line_sizes)
  129. # block_font = collections.Counter(block_fonts).most_common(1)[0][0]
  130. block_font = block_fonts.most_common(1)[0][0]
  131. block_sizes.append((block, block_size, block_font))
  132. # 计算main_text_size
  133. main_text_size = Counter(line_sizes).most_common(1)[0][0]
  134. # 计算main_text_font
  135. # main_text_font = collections.Counter(font_names).most_common(1)[0][0]
  136. # main_text_font = font_names.most_common(1)[0][0]
  137. # 删除一些可能被误识别为脚注的文本块
  138. block_sizes = [(block, block_size, block_font) for block, block_size, block_font in block_sizes if not need_remove(block)]
  139. # 检测footnote_block 并返回 footnote_bboxes
  140. # footnote_bboxes = [block['bbox'] for block, block_size, block_font in block_sizes if
  141. # block['bbox'][1] > page_height * 0.6 and block_size < main_text_size
  142. # and (len(block['lines']) < 5 or block_font != main_text_font)]
  143. # and len(block['lines']) < 5]
  144. footnote_bboxes = [block['bbox'] for block, block_size, block_font in block_sizes if
  145. block['bbox'][1] > page_height * 0.6 and
  146. sum([block_size < main_text_size,
  147. len(block['lines']) < 5,
  148. block_font != main_text_font]) >= 2]
  149. return footnote_bboxes
  150. else:
  151. return []