I am currently working on a method whose goal is to perform a closing on a binary image, but in such way that any connected components that were not connected still won't be connected after this. Hence I first extract every connected component from my input image using opencv's function cv::connectedComponents and then perform closing on temporary images with only one component, before adding these closed components in the end.
This being quite heavy, I decided to multithread this operation and to load 4 threads with a quarter of the job. The thing is that I am not able to measure any difference with the non-threaded version of the code, or by varying the number of threads. The program in which this method is used runs on a raspberry 3A+ (quad-cores).
Am I doing it wrong ? Or maybe opencv's methods already are multithreaded, preventing me to speed things up even more ?
I also tried to manually affect each thread to a dedicated core using pthread_setaffinity_np but again it did not change anything.
The code before multithreading :
void MyClass::closeConnectedComponents(const cv::Mat& src, cv::Mat& dest, const cv::InputArray& element, const int& connectivity, const int& min_area) const
{
// A label image, with a value for each connected component
cv::Mat labels;
// An image containing only a single connected component on a black background
cv::Mat current_component;
// Number of components
int nb_components;
// Compute the label image and the number of components
nb_components = cv::connectedComponents(src, labels, connectivity);
// Clear destination image
dest = cv::Mat::zeros(src.size(), src.type());
// For each component (except background)
for(int i=1; i<nb_components; i++){
// Create a mask of this component
current_component = (labels==i);
// If its area is big enough
if(cv::countNonZero(current_component) >= min_area){
// Close the current component
cv::morphologyEx(current_component, current_component, MORPH_CLOSE, element);
// Finally add it to the destination image
cv::bitwise_or(dest, current_component, dest);
}
}
// Scale destination image since for now it is only ones and zeros
dest *= 0xFF;
}
The code after multithreading :
void MyClass::closeConnectedComponentsThread(const cv::Mat& labels, cv::Mat& dest_part, const cv::InputArray& element, int min_area, int component_start_i, int component_end_i){
// An image containing only a single connected component on a black background
cv::Mat current_component;
// Process each component from component_start_i (included) to component_end_i (excluded)
for(int i = component_start_i; i < component_end_i; i++){
// Create a mask of this component
current_component = (labels==i);
// If its area is big enough
if(cv::countNonZero(current_component) >= min_area){
// Close the current component
cv::morphologyEx(current_component, current_component, MORPH_CLOSE, element);
// Finally add it to the temporary destination image
cv::bitwise_or(dest_part, current_component, dest_part);
}
}
}
void MyClass::closeConnectedComponents(const cv::Mat& src, cv::Mat& dest, const cv::InputArray& element, const int& connectivity, const int& min_area)
{
// A label image, with a value for each connected component
cv::Mat labels;
// Number of components
int nb_components;
// The number of components each individual thread will process
int range;
// A vector containing each single thread object
std::vector<std::thread> t(NB_THREADS);
// A vector containing a temporary destination image for each thread
std::vector<cv::Mat> dest_temp(NB_THREADS, cv::Mat::zeros(src.size(), src.type()));
// Compute the label image and the number of components
nb_components = cv::connectedComponents(src, labels, connectivity);
// Clear destination image
dest = cv::Mat::zeros(src.size(), src.type());
// Compute the range of each thread
range = (int)ceil((double)nb_components/NB_THREADS);
// Start the threads
for(int i=0; i<NB_THREADS; i++){
t[i] = std::thread(
&MyClass::closeConnectedComponentsThread, this,
labels, std::ref(dest_temp[i]), element, min_area, min(1+i*range, nb_components), min(1+(i+1)*range, nb_components)
);
}
// Wait for each thread to end
for(int i=0; i<NB_THREADS; i++){
if(t[i].joinable()){
t[i].join();
// Add each temporary image to the output image
cv::bitwise_or(dest, dest_temp[i], dest);
}
}
// Scale destination image since for now it is only ones and zeros
dest *= 0xFF;
}