I am currently working on a university reinforcement learning project with Tensorforce.
The setting is as follows: In a production line are 8 machines and we have a total of 40 measures that can be implemented to optimize those machines (5 measures per machine). Each measure improves the quality factor of a specific machine and has a certain cost. The goal is to select 10 measures out of the 40 that maximize the formula "Quality factor Machine 1 * Quality Factor Machine 2 * ... * Quality Factor Machine 8 - Total Costs of the 10 implemented measures" --> this is also going to be the reward function.
The goal of the reinforcement agent is to learn which sequence of measures optimizes the abovementioned function. The order here is important as the quality factors on that machine are being overriden, eg. first measure chosen updates the qualityfactor of machine 1 to 0.98 and the second measure updates it to 0.97, then the reward will be lower! Hence, the agent has to learn to implement worse measures first before updating to the best one. For 8 machines and 10 decisions, there will be machines that have to be updated twice! However, when the same measure is chosen twice, a reward of -1 is given.
I created a custom environment that looks like this below. The state is defined by the 8 quality factors of each machine and the actions are the 40 measures to choose from.
class SimulationEnvironment(Environment):
def __init__(self):
super().__init__()
self.SimulationModel = SimulationModel() # 8 machines, 40 measures, 10 decisions
self.NUM_ACTIONS = len(self.SimulationModel.actions)
self.finished = False
self.episode_end = False
self.STATES_SIZE = len(self.SimulationModel.state)
self.max_step_per_episode = 10
def states(self):
return dict(type="float", shape=(self.STATES_SIZE,))
def actions(self):
return {
"measure": dict(type="int", num_values=self.NUM_ACTIONS),
}
def max_episode_timesteps(self):
return 10
def close(self):
super().close()
def reset(self):
self.SimulationModel = SimulationModel()
state = np.array([0.8,0.88,0.88,0.88,0.88,0.88,0.88,0.88]) # initial quality factors of each machine
return state
def execute(self, actions):
reward = 0
next_state, terminal, reward = self.SimulationModel.get_nextState(actions)
return next_state, terminal, reward
The SimulationModel (where the function get_nextState is defined looks like this:
class SimulationModel:
def __init__(self):
"""
Constants
"""
self.num_machines = 8 #number of machines
self.num_measures = 40 #number of measures
self.num_decisions = 10 #number of decisions to be made
self.initial_costs = np.array([[0],[0],[0],[0],[0],[0],[0],[0]]) # Initial Costs
self.initial_quality = np.array([[0.8],[0.88],[0.88],[0.88],[0.88],[0.88],[0.88],[0.88]]) # Initial Quality Vector
self.costs = np.array([[10],[3],[50],[22],[10], # new costs for each measure, machine 1
[5],[1.5],[37.5],[20],[7], # machine 2
[5],[1.5],[37.5],[20],[6], # machine 3
[7],[2],[20],[19],[8], # machine 4
[2],[23],[19],[12],[5], # machine 5
[6],[15],[15],[35],[4], # machine 6
[2],[37.5],[36],[12],[5], # machine 7
[6],[15],[20],[37.5],[4]]) # machine 8
self.quality = np.array([[0.85],[0.92],[0.97],[0.98],[0.99], # new costs for each measure, machine 1
[0.9],[0.92],[0.97],[0.98],[0.99], # machine 2
[0.9],[0.92],[0.97],[0.98],[0.99], # machine 3
[0.9],[0.92],[0.96],[0.98],[0.99], # machine 4
[0.9],[0.96],[0.98],[0.99],[0.99], # machine 5
[0.9],[0.92],[0.94],[0.96],[0.99], # machine 6
[0.9],[0.97],[0.98],[0.99],[0.99], # machine 7
[0.9],[0.92],[0.94],[0.97],[0.99]]) # machine 8
"""
Variables
"""
self.decisions_made = 0 #Number of decisions that have been made, has to be smaller or
equal than num_decisions
"""
DYNAMICS
Machine Configurations and OEE(REWARD) will evolve at every timestep.
"""
self.configuration = np.append(self.initial_costs, self.initial_quality, axis=1)
self.OEE_previous = 32.69 #starting OEE
"""
ACTIONS:
Action vec for RL Process Times and OEE
"""
self.timestep = 0 # init timestep
self.timestep_max = self.num_decisions # Max number of timesteps per episode is number of
decisions to be taken
# Represent the actions : Choose one measure
self.actions = [0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,
21,22,23,24,25,26,27,28,28,30,31,32,33,34,35,36,37,38,39]
"""
Observations
States vec for RL. Initially, no tasks are chosen
"""
self.measurestaken = [0,0,0,0,0,0,0,0,0,0, # eg. when measure 3 is selected, index 3 will
0,0,0,0,0,0,0,0,0,0, # be set to 1
0,0,0,0,0,0,0,0,0,0,
0,0,0,0,0,0,0,0,0,0]
self.state = np.array([[0.8],[0.88],[0.88],[0.88],[0.88],[0.88],[0.88],[0.88]])
'''
Get_initial_state: Reset the environment state for the new batch of measures
'''
def get_initial_state(self):
state = np.array([[0.8],[0.88],[0.88],[0.88],[0.88],[0.88],[0.88],[0.88]])
reward = 0
return state, reward
'''
This function defines the logic behind changing the states and deciding rewards based on
the input provided by agent in terms of action.
'''
def get_nextState(self, action):
done = False
#case 1, if same task selected then penalize the agent
if self.decisions_made < self.num_decisions:
if self.measurestaken[action['measure']] == 1:
reward = -1
self.decisions_made = self.decisions_made + 1
self.OEE_previous = reward
return self.state, done, reward
else:
self.measurestaken[action['measure']] = 1
else: # 10 measures already chosen
reward = self.OEE_previous
done = True
return self.state, done, reward
# if measure is between 0 and 4, machine 1 has to be updated and so on
# costs are being added every time, quality factors is being updated
if action['measure'] < 5:
self.configuration[0][0] = self.configuration[0][0] + self.costs[action['measure']]
self.configuration[0][1] = self.quality[action['measure']]
elif action['measure'] >= 5 and action['measure'] < 10:
self.configuration[1][0] = self.configuration[1][0] + self.costs[action['measure']]
self.configuration[1][1] = self.quality[action['measure']]
elif action['measure'] >= 10 and action['measure'] < 15:
self.configuration[2][0] = self.configuration[2][0] + self.costs[action['measure']]
self.configuration[2][1] = self.quality[action['measure']]
elif action['measure'] >= 15 and action['measure'] < 20:
self.configuration[3][0] = self.configuration[3][0] + self.costs[action['measure']]
self.configuration[3][1] = self.quality[action['measure']]
elif action['measure'] >= 20 and action['measure'] < 25:
self.configuration[4][0] = self.configuration[4][0] + self.costs[action['measure']]
self.configuration[4][1] = self.quality[action['measure']]
elif action['measure'] >= 25 and action['measure'] < 30:
self.configuration[5][0] = self.configuration[5][0] + self.costs[action['measure']]
self.configuration[5][1] = self.quality[action['measure']]
elif action['measure'] >= 30 and action['measure'] < 35:
self.configuration[6][0] = self.configuration[6][0] + self.costs[action['measure']]
self.configuration[6][1] = self.quality[action['measure']]
else:
self.configuration[7][0] = self.configuration[7][0] + self.costs[action['measure']]
self.configuration[7][1] = self.quality[action['measure']]
# compute the reward (OEE)
reward = self.compute_OEE(self.configuration)
self.OEE_previous = reward
self.decisions_made = self.decisions_made + 1
self.state = self.configuration[:,1] #new state are the 8 quality factor values
return self.state,done,reward
'''
Return the OEE (reward)
'''
def compute_OEE(self, configuration):
OEE = 100 * (configuration[0][1] * configuration[1][1] * configuration[2][1] *
configuration[3][1] *
configuration[4][1] * configuration[5][1] * configuration[6][1] *
configuration[7][1]) - 0.1 * (configuration[0][0] + configuration[1][0] + configuration[2][0]
+
configuration[3][0] +
configuration[4][0] + configuration[5][0] + configuration[6][0] +
configuration[7][0])
return OEE
Now, here is the main function where the problem lies:
from env.SimulationModel import SimulationModel
from env.SimulationEnv import SimulationEnvironment
from tensorforce import Agent
def main():
# Instantiate our environment & Tensorforce Agent
environment = SimulationEnvironment()
agent = Agent.create(agent='tensorforce', environment=environment,update=64,
optimizer=dict(optimizer='adam', learning_rate=1e-3),objective='policy_gradient',
reward_estimation=dict(horizon=1))
max_reward = 0
best_state =[]
best_episode = 0
# Train for 100 episodes
for episode in range(100):
# Episode using act and observe
states = environment.reset()
terminal = False
while not terminal:
actions = agent.act(states=states)
states, terminal, reward = environment.execute(actions=actions)
agent.observe(terminal=terminal, reward=reward)
if reward > max_reward:
max_reward = reward
best_state = states
best_episode = episode
print('Episode {}: reward={} state={}'.format(episode, reward, states))
print("Best Episode: " + str(best_episode))
print("Best Reward: " + str(max_reward))
print("Best state: " + str(best_state))
print("XXXXXXXXXXXXXXXXXXXXXX")
print("XXXXXXXXXXXXXXXXXXXXXX")
print("Evaluation starts now!")
print("XXXXXXXXXXXXXXXXXXXXXX")
print("XXXXXXXXXXXXXXXXXXXXXX")
# Evaluate for 100 episodes
sum_negativerewards = 0
for evaluation in range(100):
states = environment.reset()
internals = agent.initial_internals()
terminal = False
while not terminal:
actions, internals = agent.act(
states=states, internals=internals, independent=True)
states, terminal, reward_evaluation = environment.execute(actions=actions)
#print("Reward: " + str(reward_evaluation))
if reward_evaluation == -1:
sum_negativerewards += 1
print('Evaluation {}: reward={} state={}'.format(evaluation, reward_evaluation, states))
print('Number of times with reward -100:', sum_negativerewards)
# Close agent and environment
agent.close()
environment.close()
if __name__ == "__main__":
main()
For training, I get results here that look like this:
Episode 95: reward=-1 state=[0.85 0.88 0.99 0.98 0.88 0.88 0.88 0.88]
Episode 96: reward=46.544582225592315 state=[0.99 0.88 0.88 0.88 0.88 0.99 0.98 0.94]
Episode 97: reward=34.4495764099072 state=[0.97 0.88 0.88 0.88 0.98 0.88 0.9 0.88]
Episode 98: reward=35.074430548377606 state=[0.8 0.92 0.99 0.88 0.88 0.94 0.88 0.88]
Episode 99: reward=37.04327327653888 state=[0.97 0.97 0.88 0.88 0.96 0.88 0.97 0.88]
Best Episode: 54
Best Reward: 52.52461319503872
Best state: [0.99 0.88 0.88 0.98 0.88 0.88 0.99 0.99]
[Training Output][1]
However, the evaluation part has this output, clearly showing that the same measure is being chosen over and over again:
Evaluation 1: reward=-1 state=[0.99 0.88 0.88 0.88 0.88 0.88 0.88 0.88]
Reward: 39.45888404062209
Reward: -1
Reward: -1
Reward: -1
Reward: -1
Reward: -1
Reward: -1
Reward: -1
Reward: -1
Reward: -1
Reward: -1
Evaluation 2: reward=-1 state=[0.99 0.88 0.88 0.88 0.88 0.88 0.88 0.88]
Reward: 39.45888404062209
Reward: -1
Reward: -1
Reward: -1
Reward: -1
Reward: -1
Reward: -1
Reward: -1
Reward: -1
Reward: -1
Reward: -1
Evaluation 3: reward=-1 state=[0.99 0.88 0.88 0.88 0.88 0.88 0.88 0.88]
[Evaluation Output][2]
I fear that I use the wrong agent, however if I use a ppo agent for example I get this error:
Traceback (most recent call last):
File "c:\Users\Sara\Documents\Uni\Master\Seminar Data Mining in der Produktion\Reinforcement Learning\main.py", line 92, in <module>
main()
File "c:\Users\Sara\Documents\Uni\Master\Seminar Data Mining in der Produktion\Reinforcement Learning\main.py", line 47, in main
actions = agent.act(states=states)
File "C:\Users\Sara\AppData\Local\Programs\Python\Python39\lib\site-packages\tensorforce\agents\agent.py", line 415, in act
return super().act(
File "C:\Users\Sara\AppData\Local\Programs\Python\Python39\lib\site-packages\tensorforce\agents\recorder.py", line 262, in act
actions, internals = self.fn_act(
File "C:\Users\Sara\AppData\Local\Programs\Python\Python39\lib\site-packages\tensorforce\agents\agent.py", line 462, in fn_act
actions, timesteps = self.model.act(
File "C:\Users\Sara\AppData\Local\Programs\Python\Python39\lib\site-packages\tensorforce\core\module.py", line 136, in decorated
output_args = function_graphs[str(graph_params)](*graph_args)
File "C:\Users\Sara\AppData\Local\Programs\Python\Python39\lib\site-packages\tensorflow\python\util\traceback_utils.py", line 153, in error_handler
raise e.with_traceback(filtered_tb) from None
File "C:\Users\Sara\AppData\Local\Programs\Python\Python39\lib\site-packages\tensorflow\python\eager\execute.py",
line 54, in quick_execute
tensors = pywrap_tfe.TFE_Py_Execute(ctx._handle, device_name, op_name,
tensorflow.python.framework.errors_impl.InvalidArgumentError: Graph execution error:
Detected at node 'agent/TensorScatterUpdate_2' defined at (most recent call last):
File "c:\Users\Sara\Documents\Uni\Master\Seminar Data Mining in der Produktion\Reinforcement Learning\main.py",
line 92, in <module>
main()
File "c:\Users\Sara\Documents\Uni\Master\Seminar Data Mining in der Produktion\Reinforcement Learning\main.py",
line 47, in main
actions = agent.act(states=states)
File "C:\Users\Sara\AppData\Local\Programs\Python\Python39\lib\site-packages\tensorforce\agents\agent.py", line
415, in act
return super().act(
File "C:\Users\Sara\AppData\Local\Programs\Python\Python39\lib\site-packages\tensorforce\agents\recorder.py", line 262, in act
actions, internals = self.fn_act(
File "C:\Users\Sara\AppData\Local\Programs\Python\Python39\lib\site-packages\tensorforce\agents\agent.py", line
462, in fn_act
actions, timesteps = self.model.act(
File "C:\Users\Sara\AppData\Local\Programs\Python\Python39\lib\site-packages\tensorforce\core\module.py", line 136, in decorated
output_args = function_graphs[str(graph_params)](*graph_args)
File "C:\Users\Sara\AppData\Local\Programs\Python\Python39\lib\site-packages\tensorforce\core\module.py", line 107, in function_graph
args = function(self, **kwargs.to_kwargs(), **params_kwargs)
File "C:\Users\Sara\AppData\Local\Programs\Python\Python39\lib\site-packages\tensorforce\core\models\model.py",
line 609, in act
actions, internals = self.core_act(
File "C:\Users\Sara\AppData\Local\Programs\Python\Python39\lib\site-packages\tensorforce\core\module.py", line 136, in decorated
output_args = function_graphs[str(graph_params)](*graph_args)
File "C:\Users\Sara\AppData\Local\Programs\Python\Python39\lib\site-packages\tensorforce\core\module.py", line 107, in function_graph
args = function(self, **kwargs.to_kwargs(), **params_kwargs)
File "C:\Users\Sara\AppData\Local\Programs\Python\Python39\lib\site-packages\tensorforce\core\models\tensorforce.py", line 1311, in core_act
value = tf.tensor_scatter_nd_update(tensor=buffer, indices=indices, updates=action)
Node: 'agent/TensorScatterUpdate_2'
indices[0] = [0, 5] does not index into shape [1,5]
[[{{node agent/TensorScatterUpdate_2}}]] [Op:__inference_act_1114]
I honestly really do not know what to do - it is my first time implementing Reinforcement Learning as a student and would really appreciate any kind of help! [1]: https://i.stack.imgur.com/tVP5N.png [2]: https://i.stack.imgur.com/2yxR9.png