Multi-Pose tracking with opencv kalman filter Python

Viewed 195

I'm trying to implement a multi pose tracking object in order to track each subject in video frames from multi pose estimation model output. This model gives the 'yx' coordinates of the nose of each person in the frame and I want to use them with a kalman filter to track each person in the scene frame by frame. I use an array of kalman filter, one for each pose in order to mitigate the occlusion problem. This is my implementation :

class Kalman:
    def __init__(self,
                 meas=[],
                 pred=[],
                 kalman=cv2.KalmanFilter(4, 2),
                 dt = 0.03
                 ):

        self.deltaT = []
        self.dt = dt
        self.meas = meas
        self.pred = pred
        self.kalman = kalman
        self.kalman.measurementMatrix = np.array([[1, 0, 0, 0], [0, 1, 0, 0]], np.float32)
        self.kalman.transitionMatrix = np.array([[1, 0, self.dt, 0], [0, 1, 0, self.dt], [0, 0, 1, 0], [0, 0, 0, 1]], np.float32)
        # self.kalman.processNoiseCov = np.array([[1, 0, 0, 0], [0, 1, 0, 0], [0, 0, 1, 0], [0, 0, 0, 1]], np.float32) * 0.05

        self.kalman.processNoiseCov = np.array([[(self.dt ** 4) / 4, 0, (self.dt ** 3) / 2, 0],
                   [0, (self.dt ** 4) / 4, 0, (self.dt ** 3) / 2],
                   [(self.dt ** 3) / 2, 0, self.dt ** 2, 0],
                   [0, (self.dt ** 3) / 2, 0, self.dt ** 2]], np.float32) #* 1


        # self.timex = timex

    def predict(self):
        # timeP = time.time()
        # delta = timeP - self.timex
        # self.timex = timeP
        # self.kalman.transitionMatrix[0, 2] = delta
        # self.kalman.transitionMatrix[1, 3] = delta
        tp = self.kalman.predict()  # Predict (state k+1)
        self.pred.append((int(tp[0]), int(tp[1])))
        return tp

    def correct(self, kp):
        kx, ky = kp
        mp = np.array([[np.float32(kx)], [np.float32(ky)]])
        self.meas.append((int(kx), int(ky)))
        self.kalman.correct(mp)  # Correct (state k)

    def predict_correct(self, kp):
        timeP = time.time()
        delta = timeP - self.timex
        self.timex = timeP
        self.kalman.transitionMatrix[0, 2] = delta
        self.kalman.transitionMatrix[1, 3] = delta
        # kp = pose.keypoints[0]
        kx, ky = kp
        mp = np.array([[np.float32(kx)], [np.float32(ky)]])
        self.meas.append((int(kx), int(ky)))
        self.kalman.correct(mp)  # Correct (state k)
        tp = self.kalman.predict()  # Predict (state k+1)
        self.pred.append((int(tp[0]), int(tp[1])))

    def get_pred(self):
        return self.pred


class Track:
    def __init__(self, pose, timestamp, track_id, state_track, kalman=Kalman(), unseen_counter=0):
        self.pose = pose
        self.kalman = kalman
        self.timestamp = timestamp
        self.track_id = track_id
        self.state_track = state_track
        self.unseen_counter = unseen_counter

    def unseen(self):

        if self.state_track == track_state[0]:
            self.state_track = track_state[1]
            self.unseen_counter += 1
        elif self.state_track == track_state[1]:
            if self.unseen_counter > 500:
                self.state_track = track_state['2']
            else:
                self.unseen_counter += 1

    def seen(self):
        if self.state_track == track_state[1]:
            self.unseen_counter = 0
            self.state_track = track_state[0]

    def get_state(self):
        return self.state_track

    def get_unseen_counter(self):
        return self.unseen_counter

    def predict_kalman(self):
        tp = self.kalman.predict()
        return tp

    def correct_kalman(self, keypoints):
        self.kalman.correct(keypoints)

    def distance_predict(self, pose):
        # kp = self.kalman.pred[-1]
        tp = self.predict_kalman() #pose.keypoints[0]
        distance = math.sqrt(((int(tp[0]) - pose.keypoints_norm[0][0]) ** 2) + ((int(tp[1]) - pose.keypoints_norm[0][1]) ** 2))
        return distance

    def distance(self, pose):
        tp = self.predict_correct(pose.keypoints[0]) #pose.keypoints[0]
        distance = math.sqrt(((int(tp[0]) - pose.keypoints_norm[0][0]) ** 2) + ((int(tp[1]) - pose.keypoints_norm[0][1]) ** 2))
        return distance


class Tracker:
    def __init__(self, max_tracks=18, max_age=1, min_similarity=5, color_threshold=20):
        """
        max_tracks: int,
                The maximum number of tracks that an internal tracker
                will maintain. Note that this number should be set
                larger than maxPoses. How to set this
                number requires experimentation with a given detector,
                but a good starting place is about 3 * maxPoses.
        max_age: int,
                The maximum duration of time (in milliseconds) that a
                track can exist without being linked with a new detection
                before it is removed. Set this value large if you would
                like to recover people that are not detected for long
                stretches of time (at the cost of potential false
                re-identifications).
        min_similarity: float
                New poses will only be linked with tracks if the
                similarity score exceeds this threshold.

        """
        self.max_tracks = max_tracks
        self.max_age = max_age
        self.min_similarity = min_similarity
        self.tracks = {}  # Dict of tracks, key = track_id, value = instance of class Track
        self.next_id = 1
        self.color_threshold = color_threshold

    def apply(self, poses, timestamp):
        # Filters tracks based on their age.
        self.tracks = {id: track for (id, track) in self.tracks.items() if timestamp - track.timestamp < self.max_age}
        # Sort poses by their scores from most confident to least confident
        poses = sorted(poses, key=lambda body: body.score, reverse=True)
        # Performs a greedy optimization to link detections with tracks. If incoming
        # detections are not linked with existing tracks, new tracks will be created.

        unmatched_track_indices = list(self.tracks.keys())
        unmatched_detection_indices = []

        # self.kalman_track()

        for i, pose in enumerate(poses):
            if len(unmatched_track_indices) == 0:
                unmatched_detection_indices.append(i)
                continue
            # Assign the detection to the track which produces the highest pairwise
            # similarity score, assuming the score exceeds the minimum similarity
            # threshold.
            max_track_id = -1
            max_sim = -1
            for track_id in unmatched_track_indices:

                #### NEW

                # IoU, color_distance = self.similarity(pose, self.tracks[track_id])
                # if IoU >= 0.5 and color_distance < self.color_threshold:
                #     max_track_id = track_id
                # else:
                dist = self.tracks[track_id].distance_predict(pose)
                print("La distanza predetta da kalman del track_id :" + str(track_id) + " è " + str(dist))
                if dist <= 2 and dist > max_sim:  #
                    max_track_id = track_id
                    max_sim = dist
            # if max_track_id > 0:
            #     self.tracks[max_track_id].correct_kalman(pose.keypoints_norm[0])
                ####

            # for track_id in unmatched_track_indices:
            if max_track_id >= 0: #and max_track_id == track_id
                self.tracks[max_track_id].correct_kalman(pose.keypoints_norm[0])
                self.tracks[max_track_id].seen()
                pose.track_id = max_track_id
                self.update_track(max_track_id, pose, timestamp)
                unmatched_track_indices.remove(max_track_id)
            else: # max_track_id <= 0
                unmatched_detection_indices.append(i)

        # New tracks for all unmatched detections.
        for i in unmatched_detection_indices:
            track_id = self.create_track(poses[i], timestamp, state=track_state[0]) # track_id =
            poses[i].track_id = track_id

        # # If there are too many tracks, we keep only the self.max_tracks freshest tracks
        # if len(self.tracks) > self.max_tracks:
        #     sorted_dict = sorted(self.tracks.items(), key=lambda key_value: key_value[1].timestamp, reverse=True)[
        #                   :self.max_tracks]
        #     self.tracks = {k: v for k, v in sorted_dict}

        for track_id in unmatched_track_indices:
            self.tracks[track_id].unseen()

        self.delete_tracks()

        return poses

    def delete_tracks(self):
        for track in self.tracks.items():
            if track[1].get_state() == track_state[2]:
                del self.tracks[track.track_id]

    def create_track(self, pose, timestamp, state):
        track_id = self.next_id
        self.tracks[track_id] = Track(pose, timestamp, track_id, state)
        self.next_id += 1
        return track_id

    def update_track(self, track_id, pose, timestamp):
        self.tracks[track_id].pose = pose
        self.tracks[track_id].timestamp = timestamp

My problem is that this implementation doesn't work so well. Is there someone that can help me with this implementation?

0 Answers
Related