Tesseract text recognition from document with multicoloured background

Viewed 143

I have three different types of ID cards on a different multicoloured backgrounds. For each card I need to recognise name, company, job, date of birth and social security number of the owner.

Google ID card Apple ID card IBM ID card Google v2 ID card

I'm limited to using OpenCV v3, Tesseract v4, Keras, Scikit-Learn and Python 3.6.

Tesseract 4.x is expecting dark text on light background, so I tried working with gray image, binary image, even inverted image, but I don't seem to find a solution that would work for all types of document I have.

  • I assume it would greatly help if I would be able tu crop documents from images and ignore backgrounds that way, but I'm not sure how to recognise document from a background, could contours help?
  • Other concern is how to make my code generically handle green document with white text (Google ID), white document with dark text (IBM ID) and light yellow document with gray and yellow text (Apple)? I could make three different algorithms but I would still need to decide which one to use on each image.
  • Documents are also skewed and it would help if I would be able to deskew them in pre-processing.

Any general point or code snippets for specific parts of pre-processing would help.

Here is the code I have working only with IBM cards for now:

import datetime
from enum import Enum
import cv2
import numpy as np
from PIL import Image
import sys
import pyocr
import pyocr.builders
import matplotlib
import matplotlib.pyplot as plt


class Person:
    def __init__(self, name: str = None, date_of_birth: datetime.date = None, job: str = None, ssn: str = None,
                 company: str = None):
        self.name = name
        self.date_of_birth = date_of_birth
        self.job = job
        self.ssn = ssn
        self.company = company


class Company(Enum):
    IBM = "IBM"
    APPLE = "Apple"
    GOOGLE = "Google"


def load_image(path):
    return cv2.cvtColor(cv2.imread(path), cv2.COLOR_BGR2RGB)


def image_gray(image):
    return cv2.cvtColor(image, cv2.COLOR_RGB2GRAY)


def image_bin(image_gs):
    # bin_image = image_gs > 130
    # ret, bin_image = cv2.threshold(image_gs, 200, 255, cv2.THRESH_BINARY)
    bin_image = cv2.adaptiveThreshold(image_gs, 255, cv2.ADAPTIVE_THRESH_MEAN_C, cv2.THRESH_BINARY, 15, 5)
    return bin_image


def invert(image):
    return 255 - image


def display_image(image):
    plt.imshow(image, 'gray')
    plt.show()


def is_not_blank(myString):
    if myString and myString.strip():
        # myString is not None AND myString is not empty or blank
        return True
    # myString is None OR myString is empty or blank
    return False


def extract_info_from_image(image_path: str) -> Person:
    tools = pyocr.get_available_tools()
    if len(tools) == 0:
        print("No OCR tool found")
        sys.exit(1)
    tool = tools[0]
    lang = 'eng'

    ''' pre-processing '''
    original_image = load_image(image_path)
    grey_image = image_gray(original_image)
    bin_image = image_bin(grey_image)
    inverted_image = invert(bin_image)

    person = Person()
    person.date_of_birth = datetime.datetime.today()

    ''' text recognition '''
    line_and_word_boxes = tool.image_to_string(
        Image.fromarray(test),
        lang=lang,
        builder=pyocr.builders.LineBoxBuilder(tesseract_layout=4)
    )
    for i, line in enumerate(line_and_word_boxes):
        if is_not_blank(line.content):
            print('line %d' % i)
            print(line.content, line.position)
            print('boxes')
            for box in line.word_boxes:
                print(box.content, box.position, box.confidence)
            print()

            if line.content == Company.IBM:
                person.company = line.content

            if 'DESG' in line.content:
                person.job = ' '.join(line.content.split(' ')[1:])

            if 'DOB' in line.content or 'D.O.B' in line.content:
                print(image_path)
                date = datetime.datetime.strptime(' '.join(line.content.split(' ')[1:]), '%d, %b %Y')
                if date:
                    person.date_of_birth = date
                else:
                    person.date_of_birth = datetime.datetime.today()

    if person.company == Company.IBM:
        line_with_digits = tool.image_to_string(
            Image.fromarray(test),
            lang=lang,
            builder=pyocr.builders.DigitLineBoxBuilder(tesseract_layout=1)
        )
        for i, line in enumerate(line_with_digits):
            print('line: ', line.content)
            if is_not_blank(line.content):
                print('ssn: ', line.content)
                person.ssn = line.content
                break

    return person
0 Answers
Related