Tensorforce - Agent Training with Custom Environment

Viewed 131

I am currently working on a university reinforcement learning project with Tensorforce.

The setting is as follows: In a production line are 8 machines and we have a total of 40 measures that can be implemented to optimize those machines (5 measures per machine). Each measure improves the quality factor of a specific machine and has a certain cost. The goal is to select 10 measures out of the 40 that maximize the formula "Quality factor Machine 1 * Quality Factor Machine 2 * ... * Quality Factor Machine 8 - Total Costs of the 10 implemented measures" --> this is also going to be the reward function.

The goal of the reinforcement agent is to learn which sequence of measures optimizes the abovementioned function. The order here is important as the quality factors on that machine are being overriden, eg. first measure chosen updates the qualityfactor of machine 1 to 0.98 and the second measure updates it to 0.97, then the reward will be lower! Hence, the agent has to learn to implement worse measures first before updating to the best one. For 8 machines and 10 decisions, there will be machines that have to be updated twice! However, when the same measure is chosen twice, a reward of -1 is given.

I created a custom environment that looks like this below. The state is defined by the 8 quality factors of each machine and the actions are the 40 measures to choose from.

class SimulationEnvironment(Environment):
    def __init__(self):
        super().__init__()
        self.SimulationModel = SimulationModel() # 8 machines, 40 measures, 10 decisions
        self.NUM_ACTIONS = len(self.SimulationModel.actions)
        self.finished = False
        self.episode_end = False
        self.STATES_SIZE = len(self.SimulationModel.state)
        self.max_step_per_episode = 10
    

    def states(self):
        return dict(type="float", shape=(self.STATES_SIZE,))

    def actions(self):
        return {
            "measure": dict(type="int", num_values=self.NUM_ACTIONS),
        }

    def max_episode_timesteps(self):
     return 10

    def close(self):
        super().close()

    def reset(self):
        self.SimulationModel = SimulationModel()
        state = np.array([0.8,0.88,0.88,0.88,0.88,0.88,0.88,0.88]) # initial quality factors of each machine
        return state

    def execute(self, actions):
        reward = 0
        next_state, terminal, reward = self.SimulationModel.get_nextState(actions)
        return next_state, terminal, reward

The SimulationModel (where the function get_nextState is defined looks like this:

class SimulationModel:
def __init__(self):
    """
    Constants
    """
    self.num_machines = 8 #number of machines
    self.num_measures = 40 #number of measures
    self.num_decisions = 10 #number of decisions to be made
    self.initial_costs = np.array([[0],[0],[0],[0],[0],[0],[0],[0]]) # Initial Costs
    self.initial_quality = np.array([[0.8],[0.88],[0.88],[0.88],[0.88],[0.88],[0.88],[0.88]]) # Initial Quality Vector
    self.costs = np.array([[10],[3],[50],[22],[10], # new costs for each measure, machine 1
                            [5],[1.5],[37.5],[20],[7], # machine 2
                            [5],[1.5],[37.5],[20],[6], # machine 3
                            [7],[2],[20],[19],[8], # machine 4
                            [2],[23],[19],[12],[5], # machine 5
                            [6],[15],[15],[35],[4], # machine 6
                            [2],[37.5],[36],[12],[5], # machine 7
                            [6],[15],[20],[37.5],[4]]) # machine 8
    self.quality = np.array([[0.85],[0.92],[0.97],[0.98],[0.99], # new costs for each measure, machine 1
                            [0.9],[0.92],[0.97],[0.98],[0.99], # machine 2
                            [0.9],[0.92],[0.97],[0.98],[0.99], # machine 3
                            [0.9],[0.92],[0.96],[0.98],[0.99], # machine 4
                            [0.9],[0.96],[0.98],[0.99],[0.99], # machine 5
                            [0.9],[0.92],[0.94],[0.96],[0.99], # machine 6
                            [0.9],[0.97],[0.98],[0.99],[0.99], # machine 7
                            [0.9],[0.92],[0.94],[0.97],[0.99]]) # machine 8

    """
    Variables
    """
    self.decisions_made = 0 #Number of decisions that have been made, has to be smaller or 
equal than num_decisions

    """
    DYNAMICS
    Machine Configurations and OEE(REWARD) will evolve at every timestep.
    """
    self.configuration = np.append(self.initial_costs, self.initial_quality, axis=1)
    self.OEE_previous = 32.69 #starting OEE
    
   
    """
    ACTIONS:
    Action vec for RL Process Times and OEE
    """
    self.timestep = 0 # init timestep
    self.timestep_max = self.num_decisions # Max number of timesteps per episode is number of 
 decisions to be taken

    # Represent the actions : Choose one measure
    self.actions = [0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,
                    21,22,23,24,25,26,27,28,28,30,31,32,33,34,35,36,37,38,39]
    
    """
    Observations
    States vec for RL. Initially, no tasks are chosen
    """
    self.measurestaken = [0,0,0,0,0,0,0,0,0,0, # eg. when measure 3 is selected, index 3 will 
                          0,0,0,0,0,0,0,0,0,0, # be set to 1
                          0,0,0,0,0,0,0,0,0,0,
                          0,0,0,0,0,0,0,0,0,0] 
    self.state = np.array([[0.8],[0.88],[0.88],[0.88],[0.88],[0.88],[0.88],[0.88]])

'''
Get_initial_state: Reset the environment state for the new batch of measures
'''   
def get_initial_state(self):
    state = np.array([[0.8],[0.88],[0.88],[0.88],[0.88],[0.88],[0.88],[0.88]]) 
    reward = 0
    return state, reward

'''
    This function defines the logic behind changing the states and deciding rewards based on 
the input provided by agent in terms of action.
'''
def get_nextState(self, action):
    done = False

    #case 1, if same task selected then penalize the agent
    if self.decisions_made < self.num_decisions:
        if self.measurestaken[action['measure']] == 1:
            reward = -1
            self.decisions_made = self.decisions_made + 1
            self.OEE_previous = reward
            return self.state, done, reward
        else:
            self.measurestaken[action['measure']] = 1
    else: # 10 measures already chosen
        reward = self.OEE_previous
        done = True
        return self.state, done, reward
  
    # if measure is between 0 and 4, machine 1 has to be updated and so on
    # costs are being added every time, quality factors is being updated
    if action['measure'] < 5:
        self.configuration[0][0] = self.configuration[0][0] + self.costs[action['measure']]
        self.configuration[0][1] = self.quality[action['measure']]
    elif action['measure'] >= 5 and action['measure'] < 10:
        self.configuration[1][0] = self.configuration[1][0] + self.costs[action['measure']]
        self.configuration[1][1] = self.quality[action['measure']] 
    elif action['measure'] >= 10 and action['measure'] < 15:
        self.configuration[2][0] = self.configuration[2][0] + self.costs[action['measure']]
        self.configuration[2][1] = self.quality[action['measure']] 
    elif action['measure'] >= 15 and action['measure'] < 20:
        self.configuration[3][0] = self.configuration[3][0] + self.costs[action['measure']]
        self.configuration[3][1] = self.quality[action['measure']] 
    elif action['measure'] >= 20 and action['measure'] < 25:
        self.configuration[4][0] = self.configuration[4][0] + self.costs[action['measure']]
        self.configuration[4][1] = self.quality[action['measure']] 
    elif action['measure'] >= 25 and action['measure'] < 30:
        self.configuration[5][0] = self.configuration[5][0] + self.costs[action['measure']]
        self.configuration[5][1] = self.quality[action['measure']] 
    elif action['measure'] >= 30 and action['measure'] < 35:
        self.configuration[6][0] = self.configuration[6][0] + self.costs[action['measure']]
        self.configuration[6][1] = self.quality[action['measure']] 
    else:
        self.configuration[7][0] = self.configuration[7][0] + self.costs[action['measure']]
        self.configuration[7][1] = self.quality[action['measure']]

    # compute the reward (OEE)
    reward =  self.compute_OEE(self.configuration)
    self.OEE_previous = reward
    self.decisions_made = self.decisions_made + 1
    self.state = self.configuration[:,1] #new state are the 8 quality factor values         
    return self.state,done,reward

'''
 Return the OEE (reward)
'''        
def compute_OEE(self, configuration):
    OEE = 100 * (configuration[0][1] * configuration[1][1] * configuration[2][1] * 
configuration[3][1] * 
                configuration[4][1] * configuration[5][1] * configuration[6][1] * 
configuration[7][1]) - 0.1 * (configuration[0][0] + configuration[1][0] + configuration[2][0] 
+ 
configuration[3][0] + 
                configuration[4][0] + configuration[5][0] + configuration[6][0] + 
configuration[7][0])
    return OEE

Now, here is the main function where the problem lies:

from env.SimulationModel import SimulationModel
from env.SimulationEnv import SimulationEnvironment
from tensorforce import Agent

def main():
    # Instantiate our environment & Tensorforce Agent
    environment = SimulationEnvironment()
    
    agent = Agent.create(agent='tensorforce', environment=environment,update=64, 
optimizer=dict(optimizer='adam', learning_rate=1e-3),objective='policy_gradient', 
reward_estimation=dict(horizon=1))

max_reward = 0
best_state =[]
best_episode = 0
 # Train for 100 episodes
for episode in range(100):

    # Episode using act and observe
    states = environment.reset()
    terminal = False
    while not terminal:
        actions = agent.act(states=states)
        states, terminal, reward = environment.execute(actions=actions)
        agent.observe(terminal=terminal, reward=reward)
    if reward > max_reward:
        max_reward = reward
        best_state = states
        best_episode = episode
    print('Episode {}: reward={} state={}'.format(episode, reward, states))

print("Best Episode: " + str(best_episode))
print("Best Reward: " + str(max_reward))
print("Best state: " + str(best_state))
print("XXXXXXXXXXXXXXXXXXXXXX")
print("XXXXXXXXXXXXXXXXXXXXXX")
print("Evaluation starts now!")
print("XXXXXXXXXXXXXXXXXXXXXX")
print("XXXXXXXXXXXXXXXXXXXXXX")
# Evaluate for 100 episodes
sum_negativerewards = 0
for evaluation in range(100):
    states = environment.reset()
    internals = agent.initial_internals()
    terminal = False
    while not terminal:
        actions, internals = agent.act(
            states=states, internals=internals, independent=True)
        states, terminal, reward_evaluation = environment.execute(actions=actions)
        #print("Reward: " + str(reward_evaluation))
    if reward_evaluation == -1:
        sum_negativerewards += 1
    print('Evaluation {}: reward={} state={}'.format(evaluation, reward_evaluation, states))
print('Number of times with reward -100:', sum_negativerewards)

# Close agent and environment   
agent.close()
environment.close()


if __name__ == "__main__":
        main()

For training, I get results here that look like this:

Episode 95: reward=-1 state=[0.85 0.88 0.99 0.98 0.88 0.88 0.88 0.88]
Episode 96: reward=46.544582225592315 state=[0.99 0.88 0.88 0.88 0.88 0.99 0.98 0.94]
Episode 97: reward=34.4495764099072 state=[0.97 0.88 0.88 0.88 0.98 0.88 0.9  0.88]
Episode 98: reward=35.074430548377606 state=[0.8  0.92 0.99 0.88 0.88 0.94 0.88 0.88]
Episode 99: reward=37.04327327653888 state=[0.97 0.97 0.88 0.88 0.96 0.88 0.97 0.88]
Best Episode: 54
Best Reward: 52.52461319503872
Best state: [0.99 0.88 0.88 0.98 0.88 0.88 0.99 0.99]

[Training Output][1]

However, the evaluation part has this output, clearly showing that the same measure is being chosen over and over again:

Evaluation 1: reward=-1 state=[0.99 0.88 0.88 0.88 0.88 0.88 0.88 0.88]
Reward: 39.45888404062209
Reward: -1
Reward: -1
Reward: -1
Reward: -1
Reward: -1
Reward: -1
Reward: -1
Reward: -1
Reward: -1
Reward: -1
Evaluation 2: reward=-1 state=[0.99 0.88 0.88 0.88 0.88 0.88 0.88 0.88]
Reward: 39.45888404062209
Reward: -1
Reward: -1
Reward: -1
Reward: -1
Reward: -1
Reward: -1
Reward: -1
Reward: -1
Reward: -1
Reward: -1
Evaluation 3: reward=-1 state=[0.99 0.88 0.88 0.88 0.88 0.88 0.88 0.88]

[Evaluation Output][2]

I fear that I use the wrong agent, however if I use a ppo agent for example I get this error:

Traceback (most recent call last):
File "c:\Users\Sara\Documents\Uni\Master\Seminar Data Mining in der Produktion\Reinforcement Learning\main.py", line 92, in <module>
    main()
  File "c:\Users\Sara\Documents\Uni\Master\Seminar Data Mining in der Produktion\Reinforcement Learning\main.py", line 47, in main
    actions = agent.act(states=states)
  File "C:\Users\Sara\AppData\Local\Programs\Python\Python39\lib\site-packages\tensorforce\agents\agent.py", line 415, in act
    return super().act(
  File "C:\Users\Sara\AppData\Local\Programs\Python\Python39\lib\site-packages\tensorforce\agents\recorder.py", line 262, in act
    actions, internals = self.fn_act(
  File "C:\Users\Sara\AppData\Local\Programs\Python\Python39\lib\site-packages\tensorforce\agents\agent.py", line 462, in fn_act
    actions, timesteps = self.model.act(
  File "C:\Users\Sara\AppData\Local\Programs\Python\Python39\lib\site-packages\tensorforce\core\module.py", line 136, in decorated
    output_args = function_graphs[str(graph_params)](*graph_args)
  File "C:\Users\Sara\AppData\Local\Programs\Python\Python39\lib\site-packages\tensorflow\python\util\traceback_utils.py", line 153, in error_handler
    raise e.with_traceback(filtered_tb) from None
  File "C:\Users\Sara\AppData\Local\Programs\Python\Python39\lib\site-packages\tensorflow\python\eager\execute.py", 
line 54, in quick_execute
    tensors = pywrap_tfe.TFE_Py_Execute(ctx._handle, device_name, op_name,
tensorflow.python.framework.errors_impl.InvalidArgumentError: Graph execution error:

Detected at node 'agent/TensorScatterUpdate_2' defined at (most recent call last):
    File "c:\Users\Sara\Documents\Uni\Master\Seminar Data Mining in der Produktion\Reinforcement Learning\main.py", 
line 92, in <module>
      main()
    File "c:\Users\Sara\Documents\Uni\Master\Seminar Data Mining in der Produktion\Reinforcement Learning\main.py", 
line 47, in main
      actions = agent.act(states=states)
    File "C:\Users\Sara\AppData\Local\Programs\Python\Python39\lib\site-packages\tensorforce\agents\agent.py", line 
415, in act
      return super().act(
    File "C:\Users\Sara\AppData\Local\Programs\Python\Python39\lib\site-packages\tensorforce\agents\recorder.py", line 262, in act
      actions, internals = self.fn_act(
    File "C:\Users\Sara\AppData\Local\Programs\Python\Python39\lib\site-packages\tensorforce\agents\agent.py", line 
462, in fn_act
      actions, timesteps = self.model.act(
    File "C:\Users\Sara\AppData\Local\Programs\Python\Python39\lib\site-packages\tensorforce\core\module.py", line 136, in decorated
      output_args = function_graphs[str(graph_params)](*graph_args)
    File "C:\Users\Sara\AppData\Local\Programs\Python\Python39\lib\site-packages\tensorforce\core\module.py", line 107, in function_graph
      args = function(self, **kwargs.to_kwargs(), **params_kwargs)
    File "C:\Users\Sara\AppData\Local\Programs\Python\Python39\lib\site-packages\tensorforce\core\models\model.py", 
line 609, in act
      actions, internals = self.core_act(
    File "C:\Users\Sara\AppData\Local\Programs\Python\Python39\lib\site-packages\tensorforce\core\module.py", line 136, in decorated
      output_args = function_graphs[str(graph_params)](*graph_args)
    File "C:\Users\Sara\AppData\Local\Programs\Python\Python39\lib\site-packages\tensorforce\core\module.py", line 107, in function_graph
      args = function(self, **kwargs.to_kwargs(), **params_kwargs)
    File "C:\Users\Sara\AppData\Local\Programs\Python\Python39\lib\site-packages\tensorforce\core\models\tensorforce.py", line 1311, in core_act
      value = tf.tensor_scatter_nd_update(tensor=buffer, indices=indices, updates=action)
Node: 'agent/TensorScatterUpdate_2'
indices[0] = [0, 5] does not index into shape [1,5]
         [[{{node agent/TensorScatterUpdate_2}}]] [Op:__inference_act_1114]

I honestly really do not know what to do - it is my first time implementing Reinforcement Learning as a student and would really appreciate any kind of help! [1]: https://i.stack.imgur.com/tVP5N.png [2]: https://i.stack.imgur.com/2yxR9.png

0 Answers
Related