I have trained a LSTM model in keras and now want it to deploy it through C++. I converted the .hdf5 model file to a .pb file
When I stream data through the model in C++ (one time step at a time), the model's hidden state is always reset for each time step. I confirmed this by passing the same input vector (initialized with random values) through the model and got the same output each time.
How can I retain the hidden state between calls to session->Run(...) ?
The model architecture is straight forward : 3 Dense layers, followed by an LSTM layer.
I will be using the model in streaming mode, that is passing one time step at a time as I receive data for each time step. In keras, I made the LSTM layer stateful and trained with batch_size 1. When testing the model in python, streaming works fine
If I aggregate some time steps and then pass them as a single tensor to the model in C++, it works fine. The hidden state is updated and maintained for each time step in the input tensor.
C++ Code
#include <vector>
#include <iostream>
#include <filesystem>
#include "tensorflow/core/lib/core/stringpiece.h"
#include "tensorflow/core/lib/core/status.h"
#include "tensorflow/core/lib/core/errors.h"
#include "tensorflow/core/public/session.h"
std::unique_ptr<tensorflow::Session> session;
float floatRand()
{
return float(rand()) / (float(RAND_MAX) + 1.0);
}
// loads model from .pb file
bool _loadModel(const std::string& model_file_name,
std::unique_ptr<tensorflow::Session>& session)
{
tensorflow::GraphDef graph_def;
tensorflow::Status load_model_status = ReadBinaryProto(tensorflow::Env::Default(), model_file_name, &graph_def);
if (!load_model_status.ok())
{
return false;
}
tensorflow::Status session_create_status = session->Create(graph_def);
if (!session_create_status.ok())
{
return false;
}
return true;
}
// decodes LSTM model
void _decodeLSTM(std::vector<std::vector<float>>& in_vec)
{
auto sequence_length = in_vec.size();
auto input_dim = in_vec[0].size();
tensorflow::Tensor inputs(tensorflow::DT_FLOAT, tensorflow::TensorShape({1, sequence_length, input_dim}));
auto input_tensor_mapped = inputs.tensor<float, 3>();
// copying the data into the corresponding tensor
for (int s = 0; s < sequence_length; s++)
{
for (int i = 0; i < input_dim; i++)
{
input_tensor_mapped(0, s, i) = in_vec[s][i];
}
}
std::cout << inputs.shape() << std::endl ;
std::cout << inputs.DebugString() << std::endl ;
std::string input_layer = "input";
std::string output_layer = "duration_0";
std::vector<tensorflow::Tensor> outputs;
tensorflow::Status run_status = session->Run({{input_layer, inputs}},
{output_layer}, {}, &outputs);
if (!run_status.ok())
{
std::cout<< "decodeLSTM: Failed to decode LSTM correctly!" << std::endl;
return;
}
std::cout << "Decode success" << std::endl;
std::cout << outputs[0].DebugString() << std::endl;
}
int main()
{
std::filesystem::path dur_model_path = std::filesystem::path("/media//work_dir/mymodel.pb");
session.reset(tensorflow::NewSession(tensorflow::SessionOptions()));
if (_loadModel(dur_model_path, session))
{
std::cout << "Model loaded succesfully" << std::endl;
} else {
std::cout << "Failed to load model" << std::endl;
}
// creating dummy seqeunce
std::vector<float> input_feat(657, 0.0);
for (auto &val : input_feat)
{
val = floatRand();
}
std::vector<std::vector<float>> input_seq(3, input_feat);
// only first timestep
std::vector<std::vector<float>> seqA(1);
seqA[0] = input_seq[0];
// first two timesteps
std::vector<std::vector<float>> seqB(2);
seqB[0] = input_seq[0];
seqB[1] = input_seq[1];
// all three timesteps
std::vector<std::vector<float>> seqC(3);
seqC[0] = input_seq[0];
seqC[1] = input_seq[1];
seqC[2] = input_seq[2];
std::cout << "Single timesteps :" << std::endl ;
// same sequence passed multiple times
// expecting output to change each time as
// hidden state gets updated, but not the case
std::cout << "First Pass :" << std::endl
_decodeLSTM(seqA);
std::cout << "Second Pass :" << std::endl
_decodeLSTM(seqA);
std::cout << "Third Pass :" << std::endl
_decodeLSTM(seqA);
std::cout << "Stacked timesteps :" << std::endl ;
// same sequence passed multiple times
// expecting output to change each time as
// hidden state gets updated, but not the case
std::cout << "First Pass :" << std::endl
_decodeLSTM(seqA);
std::cout << "Second Pass :" << std::endl
_decodeLSTM(seqB);
std::cout << "Third Pass :" << std::endl
_decodeLSTM(seqC);
}
Python function to convert hdf5 model file to pb model file
import os
import os.path as osp
import tensorflow as tf
from keras.models import load_model
from keras.models import model_from_json
from keras import backend as K
def convertGraph(modelPath, outdir, numoutputs, prefix, name, json):
'''
Converts an HD5F file to a .pb file for use with Tensorflow.
Args:
modelPath (str): path to the .hdf5 file
outdir (str): path to the output directory
numoutputs (int): number of model outputs
prefix (str): the prefix of the output aliasing
name (str):
Returns:
None
'''
#NOTE: If using Python > 3.2, this could be replaced with os.makedirs( name, exist_ok=True )
if not osp.isdir(outdir):
os.mkdir(outdir)
K.set_learning_phase(0)
# load the model
print("Loading from " + modelPath)
if json == True:
json_file = open(modelPath+".json", 'r')
loaded_model_json = json_file.read()
json_file.close()
net_model = model_from_json(loaded_model_json)
net_model.load_weights(modelPath+".h5")
else:
net_model = load_model(modelPath)
# Alias the outputs in the model - this sometimes makes them easier to access in TF
pred = [None]*numoutputs
pred_node_names = [None]*numoutputs
for i in range(numoutputs):
pred_node_names[i] = prefix+'_'+str(i)
pred[i] = tf.identity(net_model.output[i], name=pred_node_names[i])
print('Output nodes names are: ', pred_node_names)
sess = K.get_session()
# Write the graph in human readable
f = 'graph_def_for_reference.pb.ascii'
tf.train.write_graph(sess.graph.as_graph_def(), outdir, f, as_text=True)
print('Saved the graph definition in ascii format at: ', osp.join(outdir, f))
# Write the graph in binary .pb file
from tensorflow.python.framework import graph_util
from tensorflow.python.framework import graph_io
constant_graph = graph_util.convert_variables_to_constants(sess, sess.graph.as_graph_def(), pred_node_names)
graph_io.write_graph(constant_graph, outdir, name, as_text=False)
print('Saved the constant graph (ready for inference) at: ', osp.join(outdir, name))
I was hoping the hidden state of the LSTM layers would be automatically updated and retained after passing data through the model. This happens in python but not in C++. Each call to session->Run(...) uses the initial hidden state value