I have a project which I need to mix 2 programs into one.
Yolov4 running on tensorflow 2.x which is used to detect object in a video and Person-ReID running on tensorflow 1.x which is used to check if 2 personnes are the same or not.
After converting tensorflow 1.x into 2.x using tensorflow.compat.v1 on the majority of function, it runs perfectly.
Now comes the mix of the two programs.
#Import yolov4 library
import time
import tensorflow as tf
physical_devices = tf.config.experimental.list_physical_devices('GPU')
if len(physical_devices) > 0:
tf.config.experimental.set_memory_growth(physical_devices[0], True)
from absl import app, flags, logging
from absl.flags import FLAGS
import core.utils as utils
from core.yolov4 import filter_boxes
from tensorflow.python.saved_model import tag_constants
from PIL import Image
import cv2
import numpy as np
from imutils.video import FPS
from tensorflow.compat.v1 import ConfigProto
from tensorflow.compat.v1 import InteractiveSession
#Import reid unique library
import cuhk03_dataset
#Flag Yolov4
flags.DEFINE_string('framework', 'tf', '(tf, tflite, trt')
flags.DEFINE_string('weights', './checkpoints/yolov4-416',
'path to weights file')
flags.DEFINE_integer('size', 416, 'resize images to')
flags.DEFINE_boolean('tiny', False, 'yolo or yolo-tiny')
flags.DEFINE_string('model', 'yolov4', 'yolov3 or yolov4')
flags.DEFINE_string('video', './data/road.mp4', 'path to input video')
flags.DEFINE_float('iou', 0.45, 'iou threshold')
flags.DEFINE_float('score', 0.25, 'score threshold')
flags.DEFINE_string('output', None, 'path to output video')
flags.DEFINE_string('output_format', 'XVID', 'codec used in VideoWriter when saving video to file')
flags.DEFINE_boolean('dis_cv2_window', False, 'disable cv2 window during the process') # this is good for the .ipynb
#Flag reid
flags.DEFINE_integer('batch_size', '150', 'batch size for training')
flags.DEFINE_integer('max_steps', '1000', 'max steps for training')
flags.DEFINE_string('logs_dir', 'logs/', 'path to logs directory')
flags.DEFINE_string('data_dir', 'data/', 'path to dataset')
#Reid code
IMAGE_WIDTH = 60
IMAGE_HEIGHT = 160
image = []
#Reid preprocess
def preprocess(images, is_train):
def train():
split = tf.split(images, [1, 1])
shape = [1 for _ in list(range(split[0].get_shape()[1]))]
for i in list(range(len(split))):
split[i] = tf.reshape(split[i], [FLAGS.batch_size, IMAGE_HEIGHT, IMAGE_WIDTH, 3])
split[i] = tf.image.resize(split[i], [IMAGE_HEIGHT + 8, IMAGE_WIDTH + 3])
split[i] = tf.split(split[i], shape)
for j in list(range(len(split[i]))):
split[i][j] = tf.reshape(split[i][j], [IMAGE_HEIGHT + 8, IMAGE_WIDTH + 3, 3])
split[i][j] = tf.image.random_crop(split[i][j], [IMAGE_HEIGHT, IMAGE_WIDTH, 3])
split[i][j] = tf.image.random_flip_left_right(split[i][j])
split[i][j] = tf.image.random_brightness(split[i][j], max_delta=32. / 255.)
split[i][j] = tf.image.random_saturation(split[i][j], lower=0.5, upper=1.5)
split[i][j] = tf.image.random_hue(split[i][j], max_delta=0.2)
split[i][j] = tf.image.random_contrast(split[i][j], lower=0.5, upper=1.5)
split[i][j] = tf.image.per_image_standardization(split[i][j])
return [tf.reshape(tf.concat(split[0], axis=0), [FLAGS.batch_size, IMAGE_HEIGHT, IMAGE_WIDTH, 3]),
tf.reshape(tf.concat(split[1], axis=0), [FLAGS.batch_size, IMAGE_HEIGHT, IMAGE_WIDTH, 3])]
def val():
split = tf.split(images, [1, 1])
shape = [1 for _ in list(range(split[0].get_shape()[1]))]
for i in list(range(len(split))):
split[i] = tf.reshape(split[i], [FLAGS.batch_size, IMAGE_HEIGHT, IMAGE_WIDTH, 3])
split[i] = tf.image.resize(split[i], [IMAGE_HEIGHT, IMAGE_WIDTH])
split[i] = tf.split(split[i], shape)
for j in list(range(len(split[i]))):
split[i][j] = tf.reshape(split[i][j], [IMAGE_HEIGHT, IMAGE_WIDTH, 3])
split[i][j] = tf.image.per_image_standardization(split[i][j])
return [tf.reshape(tf.concat(split[0], axis=0), [FLAGS.batch_size, IMAGE_HEIGHT, IMAGE_WIDTH, 3]),
tf.reshape(tf.concat(split[1], axis=0), [FLAGS.batch_size, IMAGE_HEIGHT, IMAGE_WIDTH, 3])]
return tf.cond(pred=is_train, true_fn=train, false_fn=val)
#Reid network
def network(images1, images2, weight_decay):
with tf.compat.v1.variable_scope('network'):
# Tied Convolution
conv1_1 = tf.compat.v1.layers.conv2d(images1, 20, [5, 5], activation=tf.nn.relu,
kernel_regularizer=tf.keras.regularizers.l2(0.5 * (weight_decay)), name='conv1_1')
pool1_1 = tf.compat.v1.layers.max_pooling2d(conv1_1, [2, 2], [2, 2], name='pool1_1')
conv1_2 = tf.compat.v1.layers.conv2d(pool1_1, 25, [5, 5], activation=tf.nn.relu,
kernel_regularizer=tf.keras.regularizers.l2(0.5 * (weight_decay)), name='conv1_2')
pool1_2 = tf.compat.v1.layers.max_pooling2d(conv1_2, [2, 2], [2, 2], name='pool1_2')
conv2_1 = tf.compat.v1.layers.conv2d(images2, 20, [5, 5], activation=tf.nn.relu,
kernel_regularizer=tf.keras.regularizers.l2(0.5 * (weight_decay)), name='conv2_1')
pool2_1 = tf.compat.v1.layers.max_pooling2d(conv2_1, [2, 2], [2, 2], name='pool2_1')
conv2_2 = tf.compat.v1.layers.conv2d(pool2_1, 25, [5, 5], activation=tf.nn.relu,
kernel_regularizer=tf.keras.regularizers.l2(0.5 * (weight_decay)), name='conv2_2')
pool2_2 = tf.compat.v1.layers.max_pooling2d(conv2_2, [2, 2], [2, 2], name='pool2_2')
# Cross-Input Neighborhood Differences
trans = tf.transpose(a=pool1_2, perm=[0, 3, 1, 2])
shape = trans.get_shape().as_list()
m1s = tf.ones([shape[0], shape[1], shape[2], shape[3], 5, 5])
reshape = tf.reshape(trans, [shape[0], shape[1], shape[2], shape[3], 1, 1])
f = tf.multiply(reshape, m1s)
trans = tf.transpose(a=pool2_2, perm=[0, 3, 1, 2])
reshape = tf.reshape(trans, [1, shape[0], shape[1], shape[2], shape[3]])
g = []
pad = tf.pad(tensor=reshape, paddings=[[0, 0], [0, 0], [0, 0], [2, 2], [2, 2]])
for i in list(range(shape[2])):
for j in list(range(shape[3])):
g.append(pad[:,:,:,i:i+5,j:j+5])
concat = tf.concat(g, axis=0)
reshape = tf.reshape(concat, [shape[2], shape[3], shape[0], shape[1], 5, 5])
g = tf.transpose(a=reshape, perm=[2, 3, 0, 1, 4, 5])
reshape1 = tf.reshape(tf.subtract(f, g), [shape[0], shape[1], shape[2] * 5, shape[3] * 5])
reshape2 = tf.reshape(tf.subtract(g, f), [shape[0], shape[1], shape[2] * 5, shape[3] * 5])
k1 = tf.nn.relu(tf.transpose(a=reshape1, perm=[0, 2, 3, 1]), name='k1')
k2 = tf.nn.relu(tf.transpose(a=reshape2, perm=[0, 2, 3, 1]), name='k2')
# Patch Summary Features
l1 = tf.compat.v1.layers.conv2d(k1, 25, [5, 5], (5, 5), activation=tf.nn.relu,
kernel_regularizer=tf.keras.regularizers.l2(0.5 * (weight_decay)), name='l1')
l2 = tf.compat.v1.layers.conv2d(k2, 25, [5, 5], (5, 5), activation=tf.nn.relu,
kernel_regularizer=tf.keras.regularizers.l2(0.5 * (weight_decay)), name='l2')
# Across-Patch Features
m1 = tf.compat.v1.layers.conv2d(l1, 25, [3, 3], activation=tf.nn.relu,
kernel_regularizer=tf.keras.regularizers.l2(0.5 * (weight_decay)), name='m1')
pool_m1 = tf.compat.v1.layers.max_pooling2d(m1, [2, 2], [2, 2], padding='same', name='pool_m1')
m2 = tf.compat.v1.layers.conv2d(l2, 25, [3, 3], activation=tf.nn.relu,
kernel_regularizer=tf.keras.regularizers.l2(0.5 * (weight_decay)), name='m2')
pool_m2 = tf.compat.v1.layers.max_pooling2d(m2, [2, 2], [2, 2], padding='same', name='pool_m2')
# Higher-Order Relationships
concat = tf.concat([pool_m1, pool_m2], axis=3)
reshape = tf.reshape(concat, [FLAGS.batch_size, -1])
fc1 = tf.compat.v1.layers.dense(reshape, 500, tf.nn.relu, name='fc1')
fc2 = tf.compat.v1.layers.dense(fc1, 2, name='fc2')
return fc2
#Yolo main function
def yolo(_argv):
print('\nExecuting Yolo function\n')
config = ConfigProto()
config.gpu_options.allow_growth = True
session = InteractiveSession(config=config)
STRIDES, ANCHORS, NUM_CLASS, XYSCALE = utils.load_config(FLAGS)
input_size = FLAGS.size
video_path = FLAGS.video
print("Video from: ", video_path )
try:
vid = cv2.VideoCapture(int(video_path))
except:
vid = cv2.VideoCapture(video_path)
if FLAGS.framework == 'tflite':
interpreter = tf.lite.Interpreter(model_path=FLAGS.weights)
interpreter.allocate_tensors()
input_details = interpreter.get_input_details()
output_details = interpreter.get_output_details()
print(input_details)
print(output_details)
else:
saved_model_loaded = tf.saved_model.load(FLAGS.weights, tags=[tag_constants.SERVING])
infer = saved_model_loaded.signatures['serving_default']
if FLAGS.output:
# by default VideoCapture returns float instead of int
width = int(vid.get(cv2.CAP_PROP_FRAME_WIDTH))
height = int(vid.get(cv2.CAP_PROP_FRAME_HEIGHT))
fps = int(vid.get(cv2.CAP_PROP_FPS))
codec = cv2.VideoWriter_fourcc(*FLAGS.output_format)
out = cv2.VideoWriter(FLAGS.output, codec, fps, (width, height))
frame_id = 0
while True:
return_value, frame = vid.read()
if return_value:
frame = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)
image = Image.fromarray(frame)
else:
if frame_id == vid.get(cv2.CAP_PROP_FRAME_COUNT):
print("Video processing complete")
break
raise ValueError("No image! Try with another video format")
frame_size = frame.shape[:2]
image_data = cv2.resize(frame, (input_size, input_size))
image_data = image_data / 255.
image_data = image_data[np.newaxis, ...].astype(np.float32)
prev_time = time.time()
if FLAGS.framework == 'tflite':
interpreter.set_tensor(input_details[0]['index'], image_data)
interpreter.invoke()
pred = [interpreter.get_tensor(output_details[i]['index']) for i in range(len(output_details))]
if FLAGS.model == 'yolov3' and FLAGS.tiny == True:
boxes, pred_conf = filter_boxes(pred[1], pred[0], score_threshold=0.25,
input_shape=tf.constant([input_size, input_size]))
else:
boxes, pred_conf = filter_boxes(pred[0], pred[1], score_threshold=0.25,
input_shape=tf.constant([input_size, input_size]))
else:
batch_data = tf.constant(image_data)
pred_bbox = infer(batch_data)
for key, value in pred_bbox.items():
boxes = value[:, :, 0:4]
pred_conf = value[:, :, 4:]
boxes, scores, classes, valid_detections = tf.image.combined_non_max_suppression(
boxes=tf.reshape(boxes, (tf.shape(boxes)[0], -1, 1, 4)),
scores=tf.reshape(
pred_conf, (tf.shape(pred_conf)[0], -1, tf.shape(pred_conf)[-1])),
max_output_size_per_class=50,
max_total_size=50,
iou_threshold=FLAGS.iou,
score_threshold=FLAGS.score
)
pred_bbox = [boxes.numpy(), scores.numpy(), classes.numpy(), valid_detections.numpy()]
image = utils.draw_bbox(frame, pred_bbox)
curr_time = time.time()
exec_time = curr_time - prev_time
result = np.asarray(image)
info = "time: %.2f ms" %(1000*exec_time)
print(info)
result = cv2.cvtColor(image, cv2.COLOR_RGB2BGR)
if not FLAGS.dis_cv2_window:
cv2.namedWindow("result", cv2.WINDOW_AUTOSIZE)
cv2.imshow("result", result)
if cv2.waitKey(1) & 0xFF == ord('q'): break
if FLAGS.output:
out.write(result)
frame_id += 1
#Reid main
def main(args=None):
print('\nExecuting reid function\n')
FLAGS.batch_size = 1
learning_rate = tf.compat.v1.placeholder(tf.float32, name='learning_rate')
images = tf.compat.v1.placeholder(tf.float32, [2, FLAGS.batch_size, IMAGE_HEIGHT, IMAGE_WIDTH, 3], name='images')
labels = tf.compat.v1.placeholder(tf.float32, [FLAGS.batch_size, 2], name='labels')
is_train = tf.compat.v1.placeholder(tf.bool, name='is_train')
global_step = tf.Variable(0, name='global_step', trainable=False)
weight_decay = 0.0005
tarin_num_id = 0
val_num_id = 0
images1, images2 = preprocess(images, is_train)
#image3, Testimage = preprocess(images, is_train)
print('Build network')
logits = network(images1, images2, weight_decay)
loss = tf.reduce_mean(input_tensor=tf.nn.softmax_cross_entropy_with_logits(labels=tf.stop_gradient(labels), logits=logits))
inference = tf.nn.softmax(logits)
optimizer = tf.compat.v1.train.MomentumOptimizer(learning_rate, momentum=0.9)
train = optimizer.minimize(loss, global_step=global_step)
print('\nReid function finish\n')
def reid(img1,img2):
with tf.compat.v1.Session() as sess:
sess.run(tf.compat.v1.global_variables_initializer())
saver = tf.compat.v1.train.Saver()
ckpt = tf.train.get_checkpoint_state(FLAGS.logs_dir)
if ckpt and ckpt.model_checkpoint_path:
print('Restore model')
saver.restore(sess, ckpt.model_checkpoint_path)
try:
image1 = img1
image1 = cv2.resize(image1, (IMAGE_WIDTH, IMAGE_HEIGHT))
image1 = cv2.cvtColor(image1, cv2.COLOR_BGR2RGB)
image1 = np.reshape(image1, (1, IMAGE_HEIGHT, IMAGE_WIDTH, 3)).astype(float)
image2 = img2
image2 = cv2.resize(image2, (IMAGE_WIDTH, IMAGE_HEIGHT))
image2 = cv2.cvtColor(image2, cv2.COLOR_BGR2RGB)
image2 = np.reshape(image2, (1, IMAGE_HEIGHT, IMAGE_WIDTH, 3)).astype(float)
test_images = np.array([image1, image2])
feed_dict = {images: test_images, is_train: False}
prediction = sess.run(inference, feed_dict=feed_dict)
return (bool(not np.argmax(prediction[0])))
except:
print('Error')
return False
#Run all program
if __name__ == '__main__':
try:
tf.compat.v1.app.run()
app.run(yolo)
except SystemExit:
pass
My problem is that I don't know how to run both tensorflow app at the same time to avoid loading each time I use person-reID to check the identity of each person detected by Yolov4.
Thank you so much