import numpy as np
import gym
from itertools import count
env = gym.make("CartPole-v0")
max_episodes = 10000
def relu(x):
x[x < 0] = 0
return x
def relu_deriv(x):
x[x < 0] = 0
x[x > 0] = 1
return x
def softmax(x):
return np.exp(x) / np.sum(np.exp(x))
class A2C:
def __init__(self, env):
self.state_size = env.observation_space.shape[0]
self.action_size = env.action_space.n
self.params = {}
self.params["w1"] = np.random.randn(self.state_size, 32)
self.params["w2"] = np.random.randn(32, 64)
self.params["w_actor"] = np.random.randn(64, self.action_size)
self.params["w_critic"] = np.random.randn(64, 1)
self.params["b1"] = np.zeros((1, 32))
self.params["b2"] = np.zeros((1, 64))
self.params["b_actor"] = np.zeros((1, self.action_size))
self.params["b_critic"] = np.zeros((1, 1))
self.gamma = 0.99
self.actor_lr = 3e-8
self.critic_lr = 3e-8
def clear_cache(self):
self.cache = {}
self.cache["l1"] = []
self.cache["l2"] = []
self.cache["logits"] = []
self.cache["values"] = []
self.cache["rewards"] = []
self.cache["dones"] = []
self.cache["states"] = []
self.cache["actions"] = []
self.actor_loss = 0
self.critic_loss = 0
def forward(self, state, cache = True):
l1 = relu(state @ self.params["w1"] + self.params["b1"])
l2 = relu(l1 @ self.params["w2"] + self.params["b2"])
logit = l2 @ self.params["w_actor"] + self.params["b_actor"]
value = l2 @ self.params["w_critic"] + self.params["b_critic"]
if cache:
self.cache["l1"].append(l1)
self.cache["l2"].append(l2)
self.cache["logits"].append(logit)
self.cache["values"].append(value)
return logit
else: return value
def compute_advantage(self, last_value):
R = last_value
advantages = []
for t, reward in reversed(list(enumerate(self.cache["rewards"]))):
R = reward + self.gamma * R * self.cache["dones"][t]
advantage = R - self.cache["values"][t]
advantages.insert(0, advantage)
advantages = np.stack(advantages).squeeze(1)
return advantages
def take_action(self, logit):
policy = softmax(logit)
action = np.random.choice(self.action_size, p = policy.ravel())
self.cache["actions"].append(action)
return action
def update(self, last_value):
advantages = self.compute_advantage(last_value)
# turn cache lists into arrays
l1 = np.stack(self.cache["l1"]).squeeze(1)
l2 = np.stack(self.cache["l2"]).squeeze(1)
logit = np.stack(self.cache["logits"]).squeeze(1)
state = np.stack(self.cache["states"]).squeeze(1)
grads = {}
# actor backprop where L = -ln[prob(action)] * advantage
k = np.exp(logit)
c = np.array([k[i, a] for i, a in enumerate(self.cache["actions"])])[:, None]
s = np.sum(k, axis = 1)[:, None]
policy_c = c / s
self.actor_loss += np.mean(-np.log(policy_c) * advantages)
dy = -c * k / s ** 2
for i, a in enumerate(self.cache["actions"]):
dy[i, a] = c[i] * (s[i] - c[i]) / s[i] ** 2
d = -advantages / policy_c
grads["w_actor"] = l2.T @ (dy * d) * self.actor_lr
grads["b_actor"] = (dy * d).sum(axis = 0) * self.actor_lr
dy_dl2 = (dy @ self.params["w_actor"].T) * relu_deriv(l2)
grads["w2"] = l1.T @ (dy_dl2 * d) * self.actor_lr
grads["b2"] = (dy_dl2 * d).sum(axis = 0) * self.actor_lr
dy_dl1 = (dy_dl2 @ self.params["w2"].T) * relu_deriv(l1)
grads["w1"] = state.T @ (dy_dl1 * d) * self.actor_lr
grads["b1"] = (dy_dl1 * d).sum(axis = 0) * self.actor_lr
# critic backprop where L = (R - value) ** 2
self.critic_loss += np.mean(0.5 * (advantages ** 2))
dl_dy = advantages
grads["w_critic"] = l2.T @ dl_dy * self.critic_lr
grads["b_critic"] = dl_dy.sum(axis = 0) * self.critic_lr
dl_dl2 = (dl_dy @ self.params["w_critic"].T) * relu_deriv(l2)
grads["w2"] += l1.T @ dl_dl2 * self.critic_lr
grads["b2"] += dl_dl2.sum(axis = 0) * self.critic_lr
dl_dl1 = (dl_dl2 @ self.params["w2"].T) * relu_deriv(l1)
grads["w1"] += state.T @ dl_dl1 * self.critic_lr
grads["b1"] += dl_dl1.sum(axis = 0) * self.critic_lr
# update weights
for param in grads.keys():
self.params[param] -= grads[param]
agent = A2C(env)
for episode in range(max_episodes):
state = env.reset()
agent.clear_cache()
total_reward = 0
for t in count():
state = np.reshape(state, [1, agent.state_size])
logit = agent.forward(state)
action = agent.take_action(logit)
next_state, reward, done, _ = env.step(action)
agent.cache["states"].append(state)
agent.cache["rewards"].append(reward)
agent.cache["dones"].append(1 - int(done))
state = next_state
total_reward += reward
if done:
break
last_state = np.reshape(next_state, [1, agent.state_size])
last_value = agent.forward(last_state, cache = False)
agent.update(last_value)
print(t)
I tried to implement an Advantage Actor-Critic method in NumPy following this Actor-Critic PyTorch example and this REINFORCE NumPy example, except I did make it so the Actor and Critic share network for the first two layers. Unfortunately, it's not working, and I have no idea why. I do assume, however, that there's a problem in my backpropagation implementation. Is that the case, or is there a completely different problem somewhere else?