I am trying to implement a DQN agent that will find the optimal path to the terminal state in the cliff-walking environment. To do this I am using an "online" net as the q-table/q-function estimator and a target net to calculate errors via MSE. The target net is updated to match the online net's weights every 15 episodes.
My problem is that my network seems to learn that the action of going right (action 0) is the most optimal, no matter what, and therefore seems to default to this, leading to it falling to it's death over and over. The agent does not seem to care about the negative reward (-100) associated with this action at all. All other steps incur a penalty of -1, and reaching the terminal state, goal, yields a reward of 100. I have been stuck here for a while now, but I'm not getting very far. If I were to guess, I would think my error lies in either my neural net itself, or my loss calculations.
Code follows under:
import random
import gym
import numpy as np
from collections import deque
import torch
import torch.nn as nn
import torch.nn.functional as F
from dataclasses import dataclass
from typing import Any
class NeuralNet(nn.Module):
def __init__(self, state_size, action_size):
super(NeuralNet, self).__init__()
self.state_size = state_size
self.action_size = action_size
self.fc1 = nn.Linear(self.state_size, 64)
self.fc2 = nn.Linear(64, 20)
self.fc3 = nn.Linear(20, self.action_size)
#torch.nn.init.xavier_uniform_(self.fc1.weight)
#torch.nn.init.xavier_uniform_(self.fc2.weight)
#torch.nn.init.xavier_uniform_(self.fc3.weight)
def forward(self, state):
state = torch.Tensor(state)
state = F.relu(self.fc1(state))
state = F.relu(self.fc2(state))
actions = F.relu(self.fc3(state)).squeeze()
#print(actions.shape)
return actions
def update(self, net):
self.load_state_dict(net.state_dict())
def encode_vector(index: int, dim: int) -> list:
"""Encode vector as one-hot vector"""
vector_encoded = np.zeros((1, dim))
vector_encoded[0, index] = 1
return vector_encoded
if __name__ == "__main__":
# CONSTANTS
mem_size = 3000
gamma = 0.95
eps_decay = 0.999
eps_min = 0.1
learning_rate = 0.01
n_episodes = 10000
batch_size = 64
steps_for_tgt_update = 15
time_limit = 200
# Initialize environment
env = gym.make('gym_cliffwalking:cliffwalking-v0')
state_size = env.observation_space.n
action_size = env.action_space.n
# Create and init the nets that will estimate the q-function
q_net = NeuralNet(state_size, action_size)
target_net = NeuralNet(state_size, action_size)
target_net.update(q_net)
opt = torch.optim.Adam(params=q_net.parameters(), lr=learning_rate)
loss_fn = nn.MSELoss()
# Initial setup
replay_buffer = deque(maxlen=mem_size)
eps = 1.0
training = True
move_cache = [] # For visualization of the final result
step = 0
for e in range(n_episodes):
if e % steps_for_tgt_update == 0:
target_net.update(q_net)
print("Target updated")
move_cache = []
if e == len(range(n_episodes)) - 100: # Smart trick where we set the net equal to the target net after training is basically done for best end results.
training = False
state = env.reset() # Initialize env. Initial state is 0
time_taken = 0
done = False
while not done and time_taken < 75:
# One-hot encoding of vector to pass to neural net
state_encoded = encode_vector(state, state_size)
# Select action based on highest Q-value
action = -1
if eps > random.random() and training:
action = random.randrange(action_size)
#print("Random: ", action)
else:
action = torch.argmax(q_net(state_encoded)).item()
#print("Not random: ", action)
move_cache.append(action)
# Get next state resulting from actiont taken
next_state, reward, is_terminal, _ = env.step(action)
#if reward % 100 != 0: reward = -0.1
next_state_encoded = encode_vector(next_state, state_size)
# Store state, action, reward, next_state vector in replay buffer
replay_buffer.append([state, action, reward, next_state, is_terminal])
done = is_terminal
time_taken += 1
# Training step
# Only start training if we have filled mem with enough samples.
# Meaning we let the agent explore a little before going directly to memory replay
if step >= batch_size*3 and training:
opt.zero_grad()
mini_batch = np.array(random.sample(replay_buffer, batch_size))
q_pred = q_net.forward([encode_vector(int(s), state_size) for s in mini_batch[:,0]])
#print(q_pred)
q_next = target_net.forward([encode_vector(int(s), state_size) for s in mini_batch[:,3]])
#print(q_next)
best_a = torch.argmax(q_next, dim=1) #.to(torch.device('cuda:0'))
rewards = torch.Tensor(list(mini_batch[:,2]))
Q_target = q_pred.clone()
indices = np.arange(batch_size)
Q_target[indices, best_a] = rewards + gamma*torch.max(q_next)
loss = loss_fn(Q_target, q_pred)
loss.backward()
opt.step()
step += 1
print(f"Episode {e}: {time_taken} steps")
if done: print("Reached goal")
print("Move cache: ", move_cache)
if eps * eps_decay > eps_min:
eps *= eps_decay