I am trying to remove horizontal and vertical lines from a image. This image is generated from a pdf using pdf2jpg library. Upon removal of the horizontal and vertical lines this image will be fed to pytesseract to extract words and their individual co-ordinates. Here I am just extracting the full text for testing purpose. I am new to OpenCV. I have written this code by accumulating code snippets from different websites including stack overflow. The code works almost perfectly other than there are some occasional remnants of vertical lines. This remnants are confusing the tesseract and sometimes is being treated as I, 1 or |. Also it seems like number of misreads(like s is read as 5, I is read as 1 or | and vice versa) by tesseract is higher for the processed image than the original image. I think the reason for that being the font sharpness is lower than the original image that we started with. What changes can be done to this code which will remove those remnants of vertical line without affecting the font sharpness. Any suggestions or guidance in right direction will be heavily appreciated. Thanks in advance
from importlib import invalidate_caches
from pytesseract import image_to_string
#from pdf2image import convert_from_path
from pdf2jpg.pdf2jpg import convert_pdf2jpg
from PIL import Image
import sys
import cv2
import numpy
def pre_process(image):
if isinstance(image, str):
image = cv2.imread(image, cv2.IMREAD_GRAYSCALE)
else:
# image = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)
pass
#Convert the image to true black n white from grayscale
threshold, image_bin = cv2.threshold(image, 128, 255, cv2.THRESH_BINARY|cv2.THRESH_OTSU)
#Invert the image to change white to black and vice versa
image_inv = 255-image_bin
#Define kernels for horizontal and vertical lines
kernel_len = numpy.array(image).shape[1]//100
vertical_kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (1, kernel_len))
horizontal_kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (kernel_len, 1))
kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (2, 2))
#Remove anything that is not a vertical line
image_inv1 = cv2.erode(image_inv, vertical_kernel, iterations=3)
vertical_lines = cv2.dilate(image_inv1, vertical_kernel, iterations=3)
#Remove anything that is not a horizontal line
image_inv2 = cv2.erode(image_inv, horizontal_kernel, iterations=3)
horizontal_lines = cv2.dilate(image_inv2, horizontal_kernel, iterations=3)
#Add horizontal and vertical lines to get all lines
image_vh = cv2.addWeighted(vertical_lines, 0.5, horizontal_lines, 0.5, 0.0)
image_vh = cv2.erode(~image_vh, kernel, iterations=2)
threshold, image_vh = cv2.threshold(image_vh, 128, 255, cv2.THRESH_BINARY|cv2.THRESH_OTSU)
# Make a inverted copy of original grayscale image
org_img_inv = cv2.bitwise_not(image)
#Apply mask of all lines
final_image_inv = cv2.bitwise_and(org_img_inv, org_img_inv, mask=image_vh)
#Invert again to get clean image without lines
image = cv2.bitwise_not(final_image_inv)
cv2.imshow("final", image)
cv2.waitKey(0)
return image
if __name__ =="__main__":
pdf_path = sys.argv[1]
images = convert_pdf2jpg(pdf_path, "temp", dpi=100, pages="ALL")
result = ""
for image_path in images[0]["output_jpgfiles"]:
# with Image.open(image_path) as image:
# text = image_to_string(image)
# result = "\n".join((result, text))
image = pre_process(image_path)
#image = pre_process(image)
text = image_to_string(image)
result = "\n".join((result, text))
# print(result)
with open("text.txt", "w") as out:
out.write(result)
# pre_process(image_path)
# break
Please find the attached pdf, which I am using as my input pdf for the code and a snip for processed image for reference. The code can be triggered from command prompt using
python .\read_pdf_ocr.py path_to_pdf_file
Environment details:
- Python: 3.7.9
- Libraries:
- opencv-python: 4.4.0.46
- pdf2jpg: 1.0
- pytesseract: 0.3.6
- Tesseract-OCR - open source OCR engine: v5.0.0-alpha.20200328





