forked from CSEdgeOfficial/Python-Programming-Internship
-
Notifications
You must be signed in to change notification settings - Fork 0
/
Copy pathPDF_converter.py
55 lines (41 loc) · 1.81 KB
/
PDF_converter.py
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
import PyPDF2
import fitz
import os
def pdf_to_text(pdf_file, output_folder):
# Open the PDF file
with open(pdf_file, 'rb') as file:
pdf_reader = PyPDF2.PdfFileReader(file)
# Create output folder if it doesn't exist
if not os.path.exists(output_folder):
os.makedirs(output_folder)
# Extract text from each page
for page_num in range(pdf_reader.numPages):
page = pdf_reader.getPage(page_num)
text = page.extractText()
# Write extracted text to a text file
text_file_path = os.path.join(output_folder, f"page_{page_num + 1}.txt")
with open(text_file_path, 'w') as text_file:
text_file.write(text)
print("Text extraction completed. Text files saved in:", output_folder)
def pdf_to_images(pdf_file, output_folder):
# Open the PDF file
pdf_document = fitz.open(pdf_file)
# Create output folder if it doesn't exist
if not os.path.exists(output_folder):
os.makedirs(output_folder)
# Iterate through each page and save as image
for page_num in range(len(pdf_document)):
page = pdf_document[page_num]
image_path = os.path.join(output_folder, f"page_{page_num + 1}.png")
pix = page.get_pixmap()
pix.writePNG(image_path)
print("Image conversion completed. Images saved in:", output_folder)
def main():
pdf_file = 'sample.pdf' # Change this to the path of your PDF file
output_folder = 'output' # Output folder where converted files will be saved
# Convert PDF to text
pdf_to_text(pdf_file, output_folder)
# Convert PDF to images
pdf_to_images(pdf_file, output_folder)
if __name__ == "__main__":
main()