How do i implement restricted range of continuous action space in RL algorithms

Viewed 221

I have been using DDPG agent on custom gym environment which has different restrictions on different action spaces, code looks like this :

self.action_space = spaces.Box(
    low=np.array([self.constraints[channel][0] for channel in self.channels]),
    high=np.array([self.constraints[channel][1] for channel in self.channels]),
    dtype=np.float64,
)

Sample legal action space looks like array([0.44, 0.58, 1.05, 0.12]). 4 different channels and all are restricted within 20% of this value.

Agent i was using is follows (partial code for neatness):

class ActorNetwork(nn.Module):
    def __init__(self, alpha, #and other params):
        super(ActorNetwork, self).__init__()
        
        #defined the actor achitecture here

    def forward(self, state):
        x = self.fc1(state)
        x = self.bn1(x)
        x = F.relu(x)
        x = self.fc2(x)
        x = self.bn2(x)
        x = F.relu(x)
        x = T.sigmoid(self.mu(x))*1.5 #1.5 is the max value in 4 channels
        return x

class Agent(object):
    def __init__(self, alpha, #and other params ):
     
                 #initialized Actor-Critic Network

    def choose_action(self, observation):
        self.actor.eval()
        observation = T.tensor(observation, dtype=T.float).to(self.actor.device)
        mu = self.actor.forward(observation).to(self.actor.device)
        mu_prime = mu + T.tensor(np.random.normal(0.0, 0.07,size=4), #hard coded to test 
                                 dtype=T.float).to(self.actor.device)
        mu_prime = T.clamp(mu_prime, min=0.0, max = 1.5)
        self.actor.train()
        return mu_prime.cpu().detach().numpy()

                 
    def predict_next_state(self, observation):
        self.actor.eval()
        observation = T.tensor(observation, dtype=T.float).to(self.actor.device)
        mu = self.actor.forward(observation).to(self.actor.device)
        return mu.cpu().detach().numpy()

    def learn(self):
        if self.memory.mem_cntr < self.batch_size:
            return
        state, action, reward, new_state, done = \
                                      self.memory.sample_buffer(self.batch_size)

        reward = T.tensor(reward, dtype=T.float).to(self.critic.device)
        done = T.tensor(done).to(self.critic.device)
        new_state = T.tensor(new_state, dtype=T.float).to(self.critic.device)
        action = T.tensor(action, dtype=T.float).to(self.critic.device)
        state = T.tensor(state, dtype=T.float).to(self.critic.device)
                 
        self.critic.eval()
        target_actions = self.target_actor.forward(new_state)
        critic_value_ = self.target_critic.forward(new_state, target_actions)
        critic_value = self.critic.forward(state, action)
        
        #updating network with losses 

But during training, agent outputs action beyond the defined constraints in action space box env. How do i implement this so agent chooses actions only from the action space? Currently, I am penalizing every illegal action as sum of sqrt of difference from legal space. It doesn't seem to work for higher number of action space, since Agent just converges early and doesn't learn after that.

0 Answers
Related