I am trying to create a reinforcement learning algorithm which optimizes Pulltimes (timestamp) on a flight schedule, this happens by the agent substracting a number between 30-60 from the current STD (timestamp). Once it has iterrated through the entire dataframe a reward is calculated based on the bottlenecks created by these new pulltimes that have been created. The goal is to minimize the bottlenecks. So in essense I am using the Pulltime column to free up bottlenecks which happen in the STD column due to a lot of simuntaneous flights.
The reward part of the code has been created and works, however I am constantly running into error in regards to the observation space and observations.
I have a dataframe consisting of STD's and Pulltimes with the following datetime format "2022-07-27 22:00:00" which are sorted by earliest to latest timestamp.
import gym
from gym import spaces
import numpy as np
from typing import Optional
import numpy as np
from datetime import date, timedelta, time
from reward_calculation import calc_total_reward
import os
import pandas as pd
from stable_baselines3 import DQN, A2C
from stable_baselines3.common.env_checker import check_env
class PTOPTEnv(gym.Env):
def __init__(self, df):
super(PTOPTEnv, self).__init__()
self.render_mode = None # Define the attribute render_mode in your environment
self.df = df
self.df_length = len(df.index)-1
self.curr_progress = 0
self.action_space = spaces.Discrete(30)
#self.observation_space = spaces.Box(low=np.array([-np.inf]), high=np.array([np.inf]), dtype=np.int)
self.observation_space = spaces.Box(low=0, high=np.inf, shape = (5,))
#Pulltimes = self.df.loc[:, "STD"].to_numpy()
def step(self, action):
STD = self.df.loc[self.curr_progress, "STD"]
print(action, action+30)
self.df.loc[self.curr_progress, "Pulltime"] = self.df.loc[self.curr_progress, "STD"]-timedelta(minutes=action+30)
# An episode is done if the agent has reached the target
done = True if self.curr_progress==self.df_length else False
reward = 100000-calc_total_reward(self.df) if done else 0 # Binary sparse rewards
observation = self._get_obs()
info = {}
self.curr_progress += 1
return observation, reward, done, info
def reset(self):
self.curr_progress = 0
observation = self._get_obs()
info = self._get_info()
return observation
def _get_obs(self):
# Get the data points for the previous entries
frame = np.array([
self.df.loc[0: self.curr_progress, 'Pulltime'].values,
self.df.loc[:, 'Pulltime'].values,
self.df.loc[self.curr_progress: , 'Pulltime'].values,
], dtype='datetime64')
obs = np.append(frame, [[self.curr_progress, 0], [0]], axis=0)
print(obs)
print(obs.shape)
print(type(obs))
return obs
def _get_info(self):
return {"Test": 0}
dir_path = os.path.dirname(os.path.realpath(__file__))
df_use = pd.read_csv(dir_path + "\\Flight_schedule.csv", sep=";", decimal=",")
df_use["STD"] = pd.to_datetime(df_use["STD"], format='%Y-%m-%d %H:%M:%S')
df_use["Pulltime"] = 0
df_use = df_use.drop(['PAX'], axis=1)
env = PTOPTEnv(df=df_use)
check_env(env)
The issue arises when doing check_env which provides the following error: "ValueError: setting an array element with a sequence. The requested array has an inhomogeneous shape after 1 dimensions. The detected shape was (3,) + inhomogeneous part."
I've tried replacing the np.array with one consisting of 0's just to see if that would get me any further but that just throws me "AssertionError: The observation returned by reset() method must be a numpy array".
So how do I go about this, i've tried everything I could find on google but it all surrounds cartpole and other RL enviorments which have nothing to do with a pandas dataframe.