I'm trying to implement a multi pose tracking object in order to track each subject in video frames from multi pose estimation model output. This model gives the 'yx' coordinates of the nose of each person in the frame and I want to use them with a kalman filter to track each person in the scene frame by frame. I use an array of kalman filter, one for each pose in order to mitigate the occlusion problem. This is my implementation :
class Kalman:
def __init__(self,
meas=[],
pred=[],
kalman=cv2.KalmanFilter(4, 2),
dt = 0.03
):
self.deltaT = []
self.dt = dt
self.meas = meas
self.pred = pred
self.kalman = kalman
self.kalman.measurementMatrix = np.array([[1, 0, 0, 0], [0, 1, 0, 0]], np.float32)
self.kalman.transitionMatrix = np.array([[1, 0, self.dt, 0], [0, 1, 0, self.dt], [0, 0, 1, 0], [0, 0, 0, 1]], np.float32)
# self.kalman.processNoiseCov = np.array([[1, 0, 0, 0], [0, 1, 0, 0], [0, 0, 1, 0], [0, 0, 0, 1]], np.float32) * 0.05
self.kalman.processNoiseCov = np.array([[(self.dt ** 4) / 4, 0, (self.dt ** 3) / 2, 0],
[0, (self.dt ** 4) / 4, 0, (self.dt ** 3) / 2],
[(self.dt ** 3) / 2, 0, self.dt ** 2, 0],
[0, (self.dt ** 3) / 2, 0, self.dt ** 2]], np.float32) #* 1
# self.timex = timex
def predict(self):
# timeP = time.time()
# delta = timeP - self.timex
# self.timex = timeP
# self.kalman.transitionMatrix[0, 2] = delta
# self.kalman.transitionMatrix[1, 3] = delta
tp = self.kalman.predict() # Predict (state k+1)
self.pred.append((int(tp[0]), int(tp[1])))
return tp
def correct(self, kp):
kx, ky = kp
mp = np.array([[np.float32(kx)], [np.float32(ky)]])
self.meas.append((int(kx), int(ky)))
self.kalman.correct(mp) # Correct (state k)
def predict_correct(self, kp):
timeP = time.time()
delta = timeP - self.timex
self.timex = timeP
self.kalman.transitionMatrix[0, 2] = delta
self.kalman.transitionMatrix[1, 3] = delta
# kp = pose.keypoints[0]
kx, ky = kp
mp = np.array([[np.float32(kx)], [np.float32(ky)]])
self.meas.append((int(kx), int(ky)))
self.kalman.correct(mp) # Correct (state k)
tp = self.kalman.predict() # Predict (state k+1)
self.pred.append((int(tp[0]), int(tp[1])))
def get_pred(self):
return self.pred
class Track:
def __init__(self, pose, timestamp, track_id, state_track, kalman=Kalman(), unseen_counter=0):
self.pose = pose
self.kalman = kalman
self.timestamp = timestamp
self.track_id = track_id
self.state_track = state_track
self.unseen_counter = unseen_counter
def unseen(self):
if self.state_track == track_state[0]:
self.state_track = track_state[1]
self.unseen_counter += 1
elif self.state_track == track_state[1]:
if self.unseen_counter > 500:
self.state_track = track_state['2']
else:
self.unseen_counter += 1
def seen(self):
if self.state_track == track_state[1]:
self.unseen_counter = 0
self.state_track = track_state[0]
def get_state(self):
return self.state_track
def get_unseen_counter(self):
return self.unseen_counter
def predict_kalman(self):
tp = self.kalman.predict()
return tp
def correct_kalman(self, keypoints):
self.kalman.correct(keypoints)
def distance_predict(self, pose):
# kp = self.kalman.pred[-1]
tp = self.predict_kalman() #pose.keypoints[0]
distance = math.sqrt(((int(tp[0]) - pose.keypoints_norm[0][0]) ** 2) + ((int(tp[1]) - pose.keypoints_norm[0][1]) ** 2))
return distance
def distance(self, pose):
tp = self.predict_correct(pose.keypoints[0]) #pose.keypoints[0]
distance = math.sqrt(((int(tp[0]) - pose.keypoints_norm[0][0]) ** 2) + ((int(tp[1]) - pose.keypoints_norm[0][1]) ** 2))
return distance
class Tracker:
def __init__(self, max_tracks=18, max_age=1, min_similarity=5, color_threshold=20):
"""
max_tracks: int,
The maximum number of tracks that an internal tracker
will maintain. Note that this number should be set
larger than maxPoses. How to set this
number requires experimentation with a given detector,
but a good starting place is about 3 * maxPoses.
max_age: int,
The maximum duration of time (in milliseconds) that a
track can exist without being linked with a new detection
before it is removed. Set this value large if you would
like to recover people that are not detected for long
stretches of time (at the cost of potential false
re-identifications).
min_similarity: float
New poses will only be linked with tracks if the
similarity score exceeds this threshold.
"""
self.max_tracks = max_tracks
self.max_age = max_age
self.min_similarity = min_similarity
self.tracks = {} # Dict of tracks, key = track_id, value = instance of class Track
self.next_id = 1
self.color_threshold = color_threshold
def apply(self, poses, timestamp):
# Filters tracks based on their age.
self.tracks = {id: track for (id, track) in self.tracks.items() if timestamp - track.timestamp < self.max_age}
# Sort poses by their scores from most confident to least confident
poses = sorted(poses, key=lambda body: body.score, reverse=True)
# Performs a greedy optimization to link detections with tracks. If incoming
# detections are not linked with existing tracks, new tracks will be created.
unmatched_track_indices = list(self.tracks.keys())
unmatched_detection_indices = []
# self.kalman_track()
for i, pose in enumerate(poses):
if len(unmatched_track_indices) == 0:
unmatched_detection_indices.append(i)
continue
# Assign the detection to the track which produces the highest pairwise
# similarity score, assuming the score exceeds the minimum similarity
# threshold.
max_track_id = -1
max_sim = -1
for track_id in unmatched_track_indices:
#### NEW
# IoU, color_distance = self.similarity(pose, self.tracks[track_id])
# if IoU >= 0.5 and color_distance < self.color_threshold:
# max_track_id = track_id
# else:
dist = self.tracks[track_id].distance_predict(pose)
print("La distanza predetta da kalman del track_id :" + str(track_id) + " è " + str(dist))
if dist <= 2 and dist > max_sim: #
max_track_id = track_id
max_sim = dist
# if max_track_id > 0:
# self.tracks[max_track_id].correct_kalman(pose.keypoints_norm[0])
####
# for track_id in unmatched_track_indices:
if max_track_id >= 0: #and max_track_id == track_id
self.tracks[max_track_id].correct_kalman(pose.keypoints_norm[0])
self.tracks[max_track_id].seen()
pose.track_id = max_track_id
self.update_track(max_track_id, pose, timestamp)
unmatched_track_indices.remove(max_track_id)
else: # max_track_id <= 0
unmatched_detection_indices.append(i)
# New tracks for all unmatched detections.
for i in unmatched_detection_indices:
track_id = self.create_track(poses[i], timestamp, state=track_state[0]) # track_id =
poses[i].track_id = track_id
# # If there are too many tracks, we keep only the self.max_tracks freshest tracks
# if len(self.tracks) > self.max_tracks:
# sorted_dict = sorted(self.tracks.items(), key=lambda key_value: key_value[1].timestamp, reverse=True)[
# :self.max_tracks]
# self.tracks = {k: v for k, v in sorted_dict}
for track_id in unmatched_track_indices:
self.tracks[track_id].unseen()
self.delete_tracks()
return poses
def delete_tracks(self):
for track in self.tracks.items():
if track[1].get_state() == track_state[2]:
del self.tracks[track.track_id]
def create_track(self, pose, timestamp, state):
track_id = self.next_id
self.tracks[track_id] = Track(pose, timestamp, track_id, state)
self.next_id += 1
return track_id
def update_track(self, track_id, pose, timestamp):
self.tracks[track_id].pose = pose
self.tracks[track_id].timestamp = timestamp
My problem is that this implementation doesn't work so well. Is there someone that can help me with this implementation?