my below code helps me to extract text paragraphs from multiple PDF's if specific words were detected. For further analyses, i want to save the respective filename as column in my dataframe.
df['File_Name'] = file_names
only outputs the last PDF in file_names. In other words: If a triggering word was detected in the file malta_document.pdf, the dataframe still outputs malta_document_Kopie.pdf.
Note: I am new to coding.
Thank you for any help/advice
from io import StringIO
import spacy
import re
import pandas as pd
from pdfminer.converter import TextConverter
from pdfminer.layout import LAParams
from pdfminer.pdfdocument import PDFDocument
from pdfminer.pdfinterp import PDFResourceManager, PDFPageInterpreter
from pdfminer.pdfpage import PDFPage
from pdfminer.pdfparser import PDFParser
regex_pattern = r".{100}?(energy|carbon|pollution|ressource|transportation|ressources|depreciation|write-off|credited|allowances|accelerated|environment|sustainability|holidays|incentive|infringement).{100}"
file_names = 'malta_document.pdf', 'malta_document_Kopie.pdf'
output_string = StringIO()
for file_name in file_names:
with open(file_name, 'rb') as in_file:
parser = PDFParser(in_file)
doc = PDFDocument(parser)
rsrcmgr = PDFResourceManager()
device = TextConverter(rsrcmgr, output_string) #, laparams=LAParams()
interpreter = PDFPageInterpreter(rsrcmgr, device)
for page in PDFPage.create_pages(doc):
interpreter.process_page(page)
input_str = output_string.getvalue()
#print(output_string.getvalue())
lst_paragraph =[]
lst_term =[]
matches = re.finditer(regex_pattern, output_string.getvalue(), re.MULTILINE)
for matchNum, match in enumerate(matches, start=1):
lst_paragraph.append("{match}".format(matchNum = matchNum, match = match.group()))
#print ("Text paragraph to be classified: {match}".format(matchNum = matchNum, match = match.group()))
for groupNum in range(0, len(match.groups())):
groupNum = groupNum + 1
lst_term.append("{group}".format(groupNum = groupNum, group = match.group(groupNum)))
#print ("Triggering string: {group}".format(groupNum = groupNum, group = match.group(groupNum)))
#print(lst_paragraph,lst_term)
df = pd.DataFrame()
df['Triggering String'] = lst_term
df['Extracted Paragraph'] = lst_paragraph
df['File_Name'] = file_names
print (df)