现实中常常会遇到从图片或pdf中提取表格,下面我们解释如何使用python实现。
#从图片中提取表格
from img2table.document import Image
# Instantiation of the image
img = Image(src="1.jpg")
from img2table.ocr import TesseractOCR
# Instantiation of the OCR, Tesseract, which requires prior installation
languages = "eng+chi_sim"
ocr = TesseractOCR(lang=languages)
# Table identification
img_tables = img.extract_tables(ocr=ocr,borderless_tables=True)
# Result of table identification
img_tables
#从PDF中提取表格
from img2table.document import PDF
from img2table.ocr import TesseractOCR
# Instantiation of the pdf
pdf = PDF(src="11.pdf")
# Instantiation of the OCR, Tesseract, which requires prior installation
ocr = TesseractOCR(n_threads=3,lang="chi_sim")
# Table identification and extraction
pdf_tables = pdf.extract_tables(ocr=ocr,borderless_tables=True,min_confidence=99)
# We can also create an excel file with the tables
pdf.to_xlsx(dest='tables.xlsx',ocr=ocr,borderless_tables=True,min_confidence=99)
使用paddleocr精确度更高,需要安装pip install img2table[paddle]
#调用PaddleOCR---准确度更高
from img2table.ocr import PaddleOCR
ocr = PaddleOCR(lang="ch")
img = Image(src="3.png")
# Table identification
img_tables = img.extract_tables(ocr=ocr,borderless_tables=True)
# Result of table identification
img_tables
高级玩法---训练自己的识别字库:可参考下面的链接中的文章,非常详细,尝试可以。