detect_tables.py 2.7 KB

1234567891011121314151617181920212223242526272829303132333435363738394041424344454647484950515253545556575859606162
  1. from magic_pdf.libs.commons import fitz # pyMuPDF库
  2. def parse_tables(page_ID: int, page: fitz.Page, json_from_DocXchain_obj: dict):
  3. """
  4. :param page_ID: int类型,当前page在当前pdf文档中是第page_D页。
  5. :param page :fitz读取的当前页的内容
  6. :param res_dir_path: str类型,是每一个pdf文档,在当前.py文件的目录下生成一个与pdf文档同名的文件夹,res_dir_path就是文件夹的dir
  7. :param json_from_DocXchain_obj: dict类型,把pdf文档送入DocXChain模型中后,提取bbox,结果保存到pdf文档同名文件夹下的 page_ID.json文件中了。json_from_DocXchain_obj就是打开后的dict
  8. """
  9. DPI = 72 # use this resolution
  10. pix = page.get_pixmap(dpi=DPI)
  11. pageL = 0
  12. pageR = int(pix.w)
  13. pageU = 0
  14. pageD = int(pix.h)
  15. #--------- 通过json_from_DocXchain来获取 table ---------#
  16. table_bbox_from_DocXChain = []
  17. xf_json = json_from_DocXchain_obj
  18. width_from_json = xf_json['page_info']['width']
  19. height_from_json = xf_json['page_info']['height']
  20. LR_scaleRatio = width_from_json / (pageR - pageL)
  21. UD_scaleRatio = height_from_json / (pageD - pageU)
  22. for xf in xf_json['layout_dets']:
  23. # {0: 'title', 1: 'figure', 2: 'plain text', 3: 'header', 4: 'page number', 5: 'footnote', 6: 'footer', 7: 'table', 8: 'table caption', 9: 'figure caption', 10: 'equation', 11: 'full column', 12: 'sub column'}
  24. # 13: 'embedding', # 嵌入公式
  25. # 14: 'isolated'} # 单行公式
  26. L = xf['poly'][0] / LR_scaleRatio
  27. U = xf['poly'][1] / UD_scaleRatio
  28. R = xf['poly'][2] / LR_scaleRatio
  29. D = xf['poly'][5] / UD_scaleRatio
  30. # L += pageL # 有的页面,artBox偏移了。不在(0,0)
  31. # R += pageL
  32. # U += pageU
  33. # D += pageU
  34. L, R = min(L, R), max(L, R)
  35. U, D = min(U, D), max(U, D)
  36. if xf['category_id'] == 7 and xf['score'] >= 0.3:
  37. table_bbox_from_DocXChain.append((L, U, R, D))
  38. table_final_names = []
  39. table_final_bboxs = []
  40. table_ID = 0
  41. for L, U, R, D in table_bbox_from_DocXChain:
  42. # cur_table = page.get_pixmap(clip=(L,U,R,D))
  43. new_table_name = "table_{}_{}.png".format(page_ID, table_ID) # 表格name
  44. # cur_table.save(res_dir_path + '/' + new_table_name) # 把表格存出在新建的文件夹,并命名
  45. table_final_names.append(new_table_name) # 把表格的名字存在list中,方便在md中插入引用
  46. table_final_bboxs.append((L, U, R, D))
  47. table_ID += 1
  48. table_final_bboxs.sort(key = lambda LURD: (LURD[1], LURD[0]))
  49. curPage_all_table_bboxs = table_final_bboxs
  50. return curPage_all_table_bboxs