pdf_tables_to_excel.py 1.2 KB

123456789101112131415161718192021222324252627282930313233343536
  1. # -*- coding: utf-8 -*-
  2. import tabula
  3. import pandas as pd
  4. def pdf_tables_to_excel(pdf_path, excel_path, pages='77-84'):
  5. """
  6. 使用tabula提取PDF表格
  7. """
  8. try:
  9. # 读取PDF表格
  10. tables = tabula.read_pdf(pdf_path, pages=pages, multiple_tables=True, lattice=True)
  11. print("找到 {len(tables)} 个表格")
  12. with pd.ExcelWriter(excel_path, engine='openpyxl') as writer:
  13. for i, df in enumerate(tables):
  14. sheet_name = 'Table_{i + 1}'
  15. # 确保sheet名称有效
  16. sheet_name = sheet_name[:31]
  17. df.to_excel(writer, sheet_name=sheet_name, index=False)
  18. print("表格 {i + 1}: {df.shape[0]} 行 x {df.shape[1]} 列")
  19. print("导出完成: {excel_path}")
  20. except Exception as e:
  21. print("错误: {e}")
  22. # 使用示例
  23. if __name__ == "__main__":
  24. pdf_tables_to_excel(
  25. "E:\projects\通辽市一张图监督实施系统\体检成果\22成果-20240704按部下发实体地域修改后成果\通辽市2022年度城市体检评估报告.pd",
  26. "E:\projects\通辽市一张图监督实施系统\体检成果\22成果-20240704按部下发实体地域修改后成果\output.xlsx",
  27. pages='77-84')