I have been using DDPG agent on custom gym environment which has different restrictions on different action spaces, code looks like this :
self.action_space = spaces.Box(
low=np.array([self.constraints[channel][0] for channel in self.channels]),
high=np.array([self.constraints[channel][1] for channel in self.channels]),
dtype=np.float64,
)
Sample legal action space looks like array([0.44, 0.58, 1.05, 0.12]). 4 different channels and all are restricted within 20% of this value.
Agent i was using is follows (partial code for neatness):
class ActorNetwork(nn.Module):
def __init__(self, alpha, #and other params):
super(ActorNetwork, self).__init__()
#defined the actor achitecture here
def forward(self, state):
x = self.fc1(state)
x = self.bn1(x)
x = F.relu(x)
x = self.fc2(x)
x = self.bn2(x)
x = F.relu(x)
x = T.sigmoid(self.mu(x))*1.5 #1.5 is the max value in 4 channels
return x
class Agent(object):
def __init__(self, alpha, #and other params ):
#initialized Actor-Critic Network
def choose_action(self, observation):
self.actor.eval()
observation = T.tensor(observation, dtype=T.float).to(self.actor.device)
mu = self.actor.forward(observation).to(self.actor.device)
mu_prime = mu + T.tensor(np.random.normal(0.0, 0.07,size=4), #hard coded to test
dtype=T.float).to(self.actor.device)
mu_prime = T.clamp(mu_prime, min=0.0, max = 1.5)
self.actor.train()
return mu_prime.cpu().detach().numpy()
def predict_next_state(self, observation):
self.actor.eval()
observation = T.tensor(observation, dtype=T.float).to(self.actor.device)
mu = self.actor.forward(observation).to(self.actor.device)
return mu.cpu().detach().numpy()
def learn(self):
if self.memory.mem_cntr < self.batch_size:
return
state, action, reward, new_state, done = \
self.memory.sample_buffer(self.batch_size)
reward = T.tensor(reward, dtype=T.float).to(self.critic.device)
done = T.tensor(done).to(self.critic.device)
new_state = T.tensor(new_state, dtype=T.float).to(self.critic.device)
action = T.tensor(action, dtype=T.float).to(self.critic.device)
state = T.tensor(state, dtype=T.float).to(self.critic.device)
self.critic.eval()
target_actions = self.target_actor.forward(new_state)
critic_value_ = self.target_critic.forward(new_state, target_actions)
critic_value = self.critic.forward(state, action)
#updating network with losses
But during training, agent outputs action beyond the defined constraints in action space box env. How do i implement this so agent chooses actions only from the action space? Currently, I am penalizing every illegal action as sum of sqrt of difference from legal space. It doesn't seem to work for higher number of action space, since Agent just converges early and doesn't learn after that.