python pdfplumber pdfテーブル抽出用

2236 ワード

 1 import pdfplumber
 2 
 3 with pdfplumber.open('test.pdf') as pdf:
 4     #page_count = len(pdf.pages())
 5     p0 = pdf.pages[0]
 6     #     ,       ,      【 PDF        ,      “  ”】
 7     #print(p0.extract_text()) 
 8     #         ,     extract_table()      
 9     for table in p0.extract_tables(): 
10         #   table   list  ,   DataFrame          
11         for line in table:
12             print(line)
13 
14 #  ImageMagick,                 
15 #http://docs.wand-py.org/en/latest/guide/install.html#install-imagemagick-on-windows
16 #https://blog.csdn.net/blmoistawinde/article/details/82051915