I have a basic implementation of actor-critic based on keras. It works for standards open ai gym.
def __init__(self, alpha, beta, gamma=0.99, n_actions=4,
layer1_size=1024, layer2_size=512, input_dims=8):
self.gamma = gamma
self.alpha = alpha
self.beta = beta
self.input_dims = input_dims
self.fc1_dims = layer1_size
self.fc2_dims = layer2_size
self.n_actions = n_actions
self.actor, self.critic, self.policy = self.build_actor_critic_network()
self.action_space = [i for i in range(n_actions)]
def build_actor_critic_network(self):
input = Input(shape=(self.input_dims,))
delta = Input(shape=[1])
dense1 = Dense(self.fc1_dims, activation='relu')(input)
dense2 = Dense(self.fc2_dims, activation='relu')(dense1)
probs = Dense(self.n_actions, activation='softmax')(dense2)
values = Dense(1, activation='linear')(dense2)
def custom_loss(y_true, y_pred):
out = K.clip(y_pred, 1e-8, 1-1e-8)
log_lik = y_true*K.log(out)
return K.sum(-log_lik*delta)
actor = Model(inputs=[input, delta], outputs=[probs])
actor.compile(optimizer=Adam(lr=self.alpha), loss=custom_loss)
critic = Model(inputs=[input], outputs=[values])
critic.compile(optimizer=Adam(lr=self.beta), loss='mean_squared_error')
policy = Model(inputs=[input], outputs=[probs])
return actor, critic, policy
def choose_action(self, observation):
state = observation[np.newaxis, :]
probabilities = self.policy.predict(state)[0]
print('make probabilities')
action = np.random.choice(self.action_space, p=probabilities)
return action
def learn(self, state, action, reward, state_, done):
state = state[np.newaxis,:]
state_ = state_[np.newaxis,:]
critic_value_ = self.critic.predict(state_)
critic_value = self.critic.predict(state)
target = reward + self.gamma*critic_value_*(1-int(done))
delta = target - critic_value
actions = np.zeros([1, self.n_actions])
actions[np.arange(1), action] = 1
self.actor.fit([state, delta], actions, verbose=0)
self.critic.fit(state, target, verbose=0)
I want to change the observation in the choose_action method to an image, the image is a screenshot, grey scale, size=256*256. The self.predict.policy() method take as input a numpy array, tensor or tf.data dataset. When I convert the image to the tensor, I get an error: "ValueError: Input 0 of layer dense is incompatible with the layer: expected axis -1 of input shape to have value 4 but received input with shape [None, 445]"