Implementing A2C in Numpy

Viewed 150
import numpy as np
import gym
from itertools import count

env = gym.make("CartPole-v0")

max_episodes = 10000

def relu(x):

    x[x < 0] = 0

    return x

def relu_deriv(x):

    x[x < 0] = 0
    x[x > 0] = 1

    return x

def softmax(x):

    return np.exp(x) / np.sum(np.exp(x))

class A2C:

    def __init__(self, env):

        self.state_size = env.observation_space.shape[0]
        self.action_size = env.action_space.n

        self.params = {}

        self.params["w1"] = np.random.randn(self.state_size, 32)
        self.params["w2"] = np.random.randn(32, 64)
        self.params["w_actor"] = np.random.randn(64, self.action_size)
        self.params["w_critic"] = np.random.randn(64, 1)

        self.params["b1"] = np.zeros((1, 32))
        self.params["b2"] = np.zeros((1, 64))
        self.params["b_actor"] = np.zeros((1, self.action_size))
        self.params["b_critic"] = np.zeros((1, 1))

        self.gamma = 0.99
        self.actor_lr = 3e-8
        self.critic_lr = 3e-8

    def clear_cache(self):

        self.cache = {}

        self.cache["l1"] = []
        self.cache["l2"] = []
        self.cache["logits"] = []
        self.cache["values"] = []
        self.cache["rewards"] = []
        self.cache["dones"] = []
        self.cache["states"] = []
        self.cache["actions"] = []

        self.actor_loss = 0
        self.critic_loss = 0

    def forward(self, state, cache = True):

        l1 = relu(state @ self.params["w1"] + self.params["b1"])
        l2 = relu(l1 @ self.params["w2"] + self.params["b2"])

        logit = l2 @ self.params["w_actor"] + self.params["b_actor"]
        value = l2 @ self.params["w_critic"] + self.params["b_critic"]

        if cache:

            self.cache["l1"].append(l1)
            self.cache["l2"].append(l2)
            self.cache["logits"].append(logit)
            self.cache["values"].append(value)

            return logit

        else: return value

    def compute_advantage(self, last_value):

        R = last_value
        advantages = []

        for t, reward in reversed(list(enumerate(self.cache["rewards"]))):

            R = reward + self.gamma * R * self.cache["dones"][t]
            advantage = R - self.cache["values"][t]

            advantages.insert(0, advantage)

        advantages = np.stack(advantages).squeeze(1)

        return advantages

    def take_action(self, logit):

        policy = softmax(logit)
        action = np.random.choice(self.action_size, p = policy.ravel())

        self.cache["actions"].append(action)

        return action

    def update(self, last_value):

        advantages = self.compute_advantage(last_value)

        # turn cache lists into arrays

        l1 = np.stack(self.cache["l1"]).squeeze(1)
        l2 = np.stack(self.cache["l2"]).squeeze(1)
        logit = np.stack(self.cache["logits"]).squeeze(1)
        state = np.stack(self.cache["states"]).squeeze(1)

        grads = {}

        # actor backprop where L = -ln[prob(action)] * advantage

        k = np.exp(logit)
        c = np.array([k[i, a] for i, a in enumerate(self.cache["actions"])])[:, None]
        s = np.sum(k, axis = 1)[:, None]

        policy_c = c / s

        self.actor_loss += np.mean(-np.log(policy_c) * advantages)

        dy = -c * k / s ** 2

        for i, a in enumerate(self.cache["actions"]):
            dy[i, a] = c[i] * (s[i] - c[i]) / s[i] ** 2

        d = -advantages / policy_c

        grads["w_actor"] = l2.T @ (dy * d) * self.actor_lr
        grads["b_actor"] = (dy * d).sum(axis = 0) * self.actor_lr

        dy_dl2 = (dy @ self.params["w_actor"].T) * relu_deriv(l2)

        grads["w2"] = l1.T @ (dy_dl2 * d) * self.actor_lr
        grads["b2"] = (dy_dl2 * d).sum(axis = 0) * self.actor_lr

        dy_dl1 = (dy_dl2 @ self.params["w2"].T) * relu_deriv(l1)

        grads["w1"] = state.T @ (dy_dl1 * d) * self.actor_lr
        grads["b1"] = (dy_dl1 * d).sum(axis = 0) * self.actor_lr

        # critic backprop where L = (R - value) ** 2

        self.critic_loss += np.mean(0.5 * (advantages ** 2))

        dl_dy = advantages

        grads["w_critic"] = l2.T @ dl_dy * self.critic_lr
        grads["b_critic"] = dl_dy.sum(axis = 0) * self.critic_lr

        dl_dl2 = (dl_dy @ self.params["w_critic"].T) * relu_deriv(l2)

        grads["w2"] += l1.T @ dl_dl2 * self.critic_lr
        grads["b2"] += dl_dl2.sum(axis = 0) * self.critic_lr

        dl_dl1 = (dl_dl2 @ self.params["w2"].T) * relu_deriv(l1)

        grads["w1"] += state.T @ dl_dl1 * self.critic_lr
        grads["b1"] += dl_dl1.sum(axis = 0) * self.critic_lr

        # update weights

        for param in grads.keys():

            self.params[param] -= grads[param]

agent = A2C(env)

for episode in range(max_episodes):

    state = env.reset()
    agent.clear_cache()

    total_reward = 0

    for t in count():

        state = np.reshape(state, [1, agent.state_size])

        logit = agent.forward(state)
        action = agent.take_action(logit)

        next_state, reward, done, _ = env.step(action)

        agent.cache["states"].append(state)
        agent.cache["rewards"].append(reward)
        agent.cache["dones"].append(1 - int(done))

        state = next_state

        total_reward += reward

        if done:
            break

    last_state = np.reshape(next_state, [1, agent.state_size])
    last_value = agent.forward(last_state, cache = False)

    agent.update(last_value)

    print(t)

I tried to implement an Advantage Actor-Critic method in NumPy following this Actor-Critic PyTorch example and this REINFORCE NumPy example, except I did make it so the Actor and Critic share network for the first two layers. Unfortunately, it's not working, and I have no idea why. I do assume, however, that there's a problem in my backpropagation implementation. Is that the case, or is there a completely different problem somewhere else?

0 Answers
Related