OCR. Character segmentation stage

Viewed 316

I am trying to implement one of the stages of the OCR system. The character segmentation stage. The code is shown below. The code is quite simple:

  • the image is being read
  • grayscale image translation
  • image binarization
  • application of dilation operation
  • selection of contours

It is assumed that each selected contour is a symbol.

The results of the algorithm are not satisfactory. Sometimes-the characters stand out well. Sometimes only parts of characters are highlighted, sometimes several characters are highlighted. Please help with the code, I really want it to correctly highlight the characters.

UPDATE 1. I am trying to implement a character segmentation system for different fonts. It turned out that there are no universal parameters of erosion and dilation operations for different fonts

Test image:

Test image

Result of character selection 1 (Small parts of characters):

Small parts of characters

Result of character selection 2 (Big parts of characters): Double characters

Full result (All parts of characters):

Full result

import cv2
import numpy as np

def letters_extract(image_file):
    img = cv2.imread(image_file)
    gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)
    thresh = cv2.adaptiveThreshold(gray, 255, cv2.ADAPTIVE_THRESH_GAUSSIAN_C, cv2.THRESH_BINARY, 11, 2)

    img_dilate = cv2.dilate(thresh, np.ones((1, 1), np.uint8), iterations=1)
    # img_erode = cv2.erode(img_dilate, np.ones((3, 3), np.uint8), iterations=1)

    # Get contours
    contours, hierarchy = cv2.findContours(img_dilate, cv2.RETR_TREE, cv2.CHAIN_APPROX_NONE)

    letters = []
    for idx, contour in enumerate(contours):
        (x, y, w, h) = cv2.boundingRect(contour)
        if hierarchy[0][idx][3] == 0:
            letter_crop = gray[y:y + h, x:x + w]       
            letters.append(letter_crop)
            cv2.imwrite(r'D:\projects\proj\test\tnr\{}.png'.format(idx), letter_crop)

    return letters

letters_extract(r'D:\projects\proj\test\test_tnr.png')
2 Answers

Run your code (a bit modified for debugging) and it looks pretty good (I've only changed the dilation mask):

import cv2
import numpy as np
import matplotlib.pyplot as plt


def letters_extract(image_file):
    img = cv2.imread(image_file)
    gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)
    thresh = cv2.adaptiveThreshold(gray, 255, cv2.ADAPTIVE_THRESH_GAUSSIAN_C, cv2.THRESH_BINARY, 11, 2)
    plt.figure(figsize=(20, 20))
    plt.imshow(thresh)
    plt.show()

    img_dilate = cv2.erode(thresh, np.ones((2,1), np.uint8))
    plt.figure(figsize=(20, 20))
    plt.imshow(img_dilate)
    plt.show()

    contours, hierarchy = cv2.findContours(img_dilate, cv2.RETR_TREE, cv2.CHAIN_APPROX_NONE)

    im_with_aabb = img.copy()
    for idx, contour in enumerate(contours):
        (x, y, w, h) = cv2.boundingRect(contour)
        if hierarchy[0][idx][3] == 0:     
            color = (255, 0, 0)
            thickness = 1
            im_with_aabb = cv2.rectangle(im_with_aabb, (x,y), (x+w,y+h), color, thickness)

    return im_with_aabb

im_with_aabb = letters_extract('test.png')
plt.figure(figsize=(20, 20))
plt.imshow(im_with_aabb)
plt.show()

But there are problems with several chars still. If your input images looks this good (no high variability between the same char in different places) I can suggest perhaps tamplate matching with each char as template.

If the data is with high variability maybe you should use a pretrained NN like tesseract.

If your data is always as clear as the image you have shared, you do not have to do dilation or erosion. I set threshold to 190 and inverse the gray image with cv2.THRESH_BINARY_INV parameter such that countours will be find around the letters. Finally, I change contour search algorithm to find only external contours with cv2.RETR_EXTERNAL parameter.

import cv2
import numpy as np

def letters_extract(image_file):
    img = cv2.imread(image_file)
    gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)
    _, thresh = cv2.threshold(gray, 190, 255, cv2.THRESH_BINARY_INV)

    contours, _ = cv2.findContours(thresh, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_NONE)

    letters = []
    for idx, contour in enumerate(contours):
        (x, y, w, h) = cv2.boundingRect(contour)
        letter_crop = gray[y:y + h, x:x + w]       
        letters.append(letter_crop)
        cv2.rectangle(img, (x,y), (x + w, y + h), (0,0,255))
    
    cv2.namedWindow("win", cv2.WINDOW_FREERATIO)
    cv2.imshow("win",img)
    cv2.waitKey()

    return letters

letters_extract('text.png')

Final image is as follows:

Modified text image

Related