Python: PDF
Revision as of 05:26, 25 October 2018 by Onnowpurbo (talk | contribs)
pyPDF2
#install pyPDF2
pip install PyPDF2
# importing all the required modules
import PyPDF2
# creating an object
file = open('example.pdf', 'rb')
# creating a pdf reader object
fileReader = PyPDF2.PdfFileReader(file)
# print the number of pages in pdf file
print(fileReader.numPages)
textract
pip install textract
# for read pdf
import textract
text = textract.process('path/to/pdf/file', method='pdfminer')