# -*- coding: utf-8 -*- import tabula import pandas as pd def pdf_tables_to_excel(pdf_path, excel_path, pages='77-84'): """ 使用tabula提取PDF表格 """ try: # 读取PDF表格 tables = tabula.read_pdf(pdf_path, pages=pages, multiple_tables=True, lattice=True) print("找到 {len(tables)} 个表格") with pd.ExcelWriter(excel_path, engine='openpyxl') as writer: for i, df in enumerate(tables): sheet_name = 'Table_{i + 1}' # 确保sheet名称有效 sheet_name = sheet_name[:31] df.to_excel(writer, sheet_name=sheet_name, index=False) print("表格 {i + 1}: {df.shape[0]} 行 x {df.shape[1]} 列") print("导出完成: {excel_path}") except Exception as e: print("错误: {e}") # 使用示例 if __name__ == "__main__": pdf_tables_to_excel( "E:\projects\通辽市一张图监督实施系统\体检成果\22成果-20240704按部下发实体地域修改后成果\通辽市2022年度城市体检评估报告.pd", "E:\projects\通辽市一张图监督实施系统\体检成果\22成果-20240704按部下发实体地域修改后成果\output.xlsx", pages='77-84')