I have three different types of ID cards on a different multicoloured backgrounds. For each card I need to recognise name, company, job, date of birth and social security number of the owner.
I'm limited to using OpenCV v3, Tesseract v4, Keras, Scikit-Learn and Python 3.6.
Tesseract 4.x is expecting dark text on light background, so I tried working with gray image, binary image, even inverted image, but I don't seem to find a solution that would work for all types of document I have.
- I assume it would greatly help if I would be able tu crop documents from images and ignore backgrounds that way, but I'm not sure how to recognise document from a background, could contours help?
- Other concern is how to make my code generically handle green document with white text (Google ID), white document with dark text (IBM ID) and light yellow document with gray and yellow text (Apple)? I could make three different algorithms but I would still need to decide which one to use on each image.
- Documents are also skewed and it would help if I would be able to deskew them in pre-processing.
Any general point or code snippets for specific parts of pre-processing would help.
Here is the code I have working only with IBM cards for now:
import datetime
from enum import Enum
import cv2
import numpy as np
from PIL import Image
import sys
import pyocr
import pyocr.builders
import matplotlib
import matplotlib.pyplot as plt
class Person:
def __init__(self, name: str = None, date_of_birth: datetime.date = None, job: str = None, ssn: str = None,
company: str = None):
self.name = name
self.date_of_birth = date_of_birth
self.job = job
self.ssn = ssn
self.company = company
class Company(Enum):
IBM = "IBM"
APPLE = "Apple"
GOOGLE = "Google"
def load_image(path):
return cv2.cvtColor(cv2.imread(path), cv2.COLOR_BGR2RGB)
def image_gray(image):
return cv2.cvtColor(image, cv2.COLOR_RGB2GRAY)
def image_bin(image_gs):
# bin_image = image_gs > 130
# ret, bin_image = cv2.threshold(image_gs, 200, 255, cv2.THRESH_BINARY)
bin_image = cv2.adaptiveThreshold(image_gs, 255, cv2.ADAPTIVE_THRESH_MEAN_C, cv2.THRESH_BINARY, 15, 5)
return bin_image
def invert(image):
return 255 - image
def display_image(image):
plt.imshow(image, 'gray')
plt.show()
def is_not_blank(myString):
if myString and myString.strip():
# myString is not None AND myString is not empty or blank
return True
# myString is None OR myString is empty or blank
return False
def extract_info_from_image(image_path: str) -> Person:
tools = pyocr.get_available_tools()
if len(tools) == 0:
print("No OCR tool found")
sys.exit(1)
tool = tools[0]
lang = 'eng'
''' pre-processing '''
original_image = load_image(image_path)
grey_image = image_gray(original_image)
bin_image = image_bin(grey_image)
inverted_image = invert(bin_image)
person = Person()
person.date_of_birth = datetime.datetime.today()
''' text recognition '''
line_and_word_boxes = tool.image_to_string(
Image.fromarray(test),
lang=lang,
builder=pyocr.builders.LineBoxBuilder(tesseract_layout=4)
)
for i, line in enumerate(line_and_word_boxes):
if is_not_blank(line.content):
print('line %d' % i)
print(line.content, line.position)
print('boxes')
for box in line.word_boxes:
print(box.content, box.position, box.confidence)
print()
if line.content == Company.IBM:
person.company = line.content
if 'DESG' in line.content:
person.job = ' '.join(line.content.split(' ')[1:])
if 'DOB' in line.content or 'D.O.B' in line.content:
print(image_path)
date = datetime.datetime.strptime(' '.join(line.content.split(' ')[1:]), '%d, %b %Y')
if date:
person.date_of_birth = date
else:
person.date_of_birth = datetime.datetime.today()
if person.company == Company.IBM:
line_with_digits = tool.image_to_string(
Image.fromarray(test),
lang=lang,
builder=pyocr.builders.DigitLineBoxBuilder(tesseract_layout=1)
)
for i, line in enumerate(line_with_digits):
print('line: ', line.content)
if is_not_blank(line.content):
print('ssn: ', line.content)
person.ssn = line.content
break
return person



