DQN network starts to predict only zeroes as q-values after x iterations in cliffwalking environment

Viewed 34

I am trying to implement a DQN agent that will find the optimal path to the terminal state in the cliff-walking environment. To do this I am using an "online" net as the q-table/q-function estimator and a target net to calculate errors via MSE. The target net is updated to match the online net's weights every 15 episodes.

My problem is that my network seems to learn that the action of going right (action 0) is the most optimal, no matter what, and therefore seems to default to this, leading to it falling to it's death over and over. The agent does not seem to care about the negative reward (-100) associated with this action at all. All other steps incur a penalty of -1, and reaching the terminal state, goal, yields a reward of 100. I have been stuck here for a while now, but I'm not getting very far. If I were to guess, I would think my error lies in either my neural net itself, or my loss calculations.

Code follows under:

import random
import gym
import numpy as np
from collections import deque
import torch
import torch.nn as nn
import torch.nn.functional as F
from dataclasses import dataclass
from typing import Any

class NeuralNet(nn.Module):
    def __init__(self, state_size, action_size):
        super(NeuralNet, self).__init__()
        self.state_size = state_size
        self.action_size = action_size
        
        self.fc1 = nn.Linear(self.state_size, 64)
        self.fc2 = nn.Linear(64, 20)
        self.fc3 = nn.Linear(20, self.action_size)
        #torch.nn.init.xavier_uniform_(self.fc1.weight)
        #torch.nn.init.xavier_uniform_(self.fc2.weight)
        #torch.nn.init.xavier_uniform_(self.fc3.weight)

    def forward(self, state):
        state = torch.Tensor(state)

        state = F.relu(self.fc1(state))
        state = F.relu(self.fc2(state))
        actions = F.relu(self.fc3(state)).squeeze()
        #print(actions.shape)
        return actions

    def update(self, net):
        self.load_state_dict(net.state_dict())



def encode_vector(index: int, dim: int) -> list:
    """Encode vector as one-hot vector"""
    vector_encoded = np.zeros((1, dim))
    vector_encoded[0, index] = 1

    return vector_encoded

if __name__ == "__main__":
    # CONSTANTS
    mem_size = 3000
    gamma = 0.95
    eps_decay = 0.999
    eps_min = 0.1
    learning_rate = 0.01
    n_episodes = 10000
    batch_size = 64
    steps_for_tgt_update = 15
    time_limit = 200

    # Initialize environment
    env = gym.make('gym_cliffwalking:cliffwalking-v0')
    state_size = env.observation_space.n
    action_size = env.action_space.n

    # Create and init the nets that will estimate the q-function
    q_net = NeuralNet(state_size, action_size)
    target_net = NeuralNet(state_size, action_size)
    target_net.update(q_net)
    opt = torch.optim.Adam(params=q_net.parameters(), lr=learning_rate)
    loss_fn = nn.MSELoss()

    # Initial setup
    replay_buffer = deque(maxlen=mem_size)
    eps = 1.0
    training = True
    move_cache = [] # For visualization of the final result
    step = 0

    for e in range(n_episodes):

        if e % steps_for_tgt_update == 0:
            target_net.update(q_net)
            print("Target updated")

        move_cache = []
        if e == len(range(n_episodes)) - 100: # Smart trick where we set the net equal to the target net after training is basically done for best end results.
            training = False
        
        state = env.reset() # Initialize env. Initial state is 0

        time_taken = 0
        done = False
        while not done and time_taken < 75:
            
            # One-hot encoding of vector to pass to neural net
            state_encoded = encode_vector(state, state_size)
            
            # Select action based on highest Q-value
            action = -1
            if eps > random.random() and training:
                action = random.randrange(action_size)
                #print("Random: ", action)
            else:
                action = torch.argmax(q_net(state_encoded)).item()
                #print("Not random: ", action)
            
            move_cache.append(action)

            # Get next state resulting from actiont taken
            next_state, reward, is_terminal, _ = env.step(action)
            #if reward % 100 != 0: reward = -0.1
            next_state_encoded = encode_vector(next_state, state_size)

            # Store state, action, reward, next_state vector in replay buffer
            replay_buffer.append([state, action, reward, next_state, is_terminal])
            done = is_terminal
            time_taken += 1

            # Training step
            # Only start training if we have filled mem with enough samples.
            # Meaning we let the agent explore a little before going directly to memory replay
            if step >= batch_size*3 and training:
                opt.zero_grad()
                
                mini_batch = np.array(random.sample(replay_buffer, batch_size))

                q_pred = q_net.forward([encode_vector(int(s), state_size) for s in mini_batch[:,0]])
                #print(q_pred)
                q_next = target_net.forward([encode_vector(int(s), state_size) for s in mini_batch[:,3]])
                #print(q_next)
                best_a = torch.argmax(q_next, dim=1) #.to(torch.device('cuda:0'))
                rewards = torch.Tensor(list(mini_batch[:,2]))

                Q_target = q_pred.clone()
                indices = np.arange(batch_size)
                Q_target[indices, best_a] = rewards + gamma*torch.max(q_next)

                loss = loss_fn(Q_target, q_pred)
                loss.backward()
                opt.step()
            
            step += 1
            
        print(f"Episode {e}: {time_taken} steps")
        if done: print("Reached goal")
        print("Move cache: ", move_cache)
        if eps * eps_decay > eps_min:
            eps *= eps_decay
0 Answers
Related