I have multiple tfrecord files named: Train_DE_01.tfrecords through Train_DE_34.tfrecords; and Devel_DE_01.tfrecords through Devel_DE_14.tfrecords. Hence, I have a training and a validation dataset. And My aim was to iterator over the examples of the tfrecords such that I retrieve 2 examples from Train_DE_01.tfrecords, 2 from Train_DE_02.tfrecords ... and 2 Train_DE_34.tfrecords. In other words, when the batch size is 68, I need 2 examples from each tfrecord file. I my code, I have used an initializable Iterator as follows:
# file_name: This is a place_holder that will contain the name of the files of the tfrecords.
def load_sewa_data(file_name, batch_size):
with tf.name_scope('sewa_tf_records'):
dataset = tf.data.TFRecordDataset(file_name).map(_parse_sewa_example).batch(batch_size)
iterator = dataset.make_initializable_iterator(shared_name='sewa_iterator')
next_batch = iterator.get_next()
names, detected, arousal, valence, liking, istalkings, images = next_batch
print(names, detected, arousal, valence, liking, istalkings, images)
return names, detected, arousal, valence, liking, istalkings, images, iterator
After running the names through a session using sess.run(); I figured out that the first 68 example are being fetched from Train_DE_01.tfrecords; then, subsequent examples are fetched from the same tfrecord until all the examples in the Train_DE_01.tfrecords are being consumed.
I have tried using the zip() function of Dataset api with the reinitializable iterator as follows:
def load_devel_sewa_tfrecords(filenames_dev, test_batch_size):
datasets_dev_iterators = []
with tf.name_scope('TFRecordsDevel'):
for file_name in filenames_dev:
dataset_dev = tf.data.TFRecordDataset(file_name).map(_parse_devel_function).batch(test_batch_size)
datasets_dev_iterators.append(dataset_dev)
dataset_dev_all = tf.data.Dataset.zip(tuple(datasets_dev_iterators))
return dataset_dev_all
def load_train_sewa_tfrecords(filenames_train, train_batch_size):
datasets_train_iterators = []
with tf.name_scope('TFRecordsTrain'):
for file_name in filenames_train:
dataset_train = tf.data.TFRecordDataset(file_name).map(_parse_train_function).batch(train_batch_size)
datasets_train_iterators.append(dataset_train)
dataset_train_all = tf.data.Dataset.zip(tuple(datasets_train_iterators))
return dataset_train_all
def load_sewa_dataset(filenames_train, train_batch_size, filenames_dev, test_batch_size):
dataset_train_all = load_train_sewa_tfrecords(filenames_train, train_batch_size)
dataset_dev_all = load_devel_sewa_tfrecords(filenames_dev, test_batch_size)
iterator = tf.data.Iterator.from_structure(dataset_train_all.output_types,
dataset_train_all.output_shapes)
training_init_op = iterator.make_initializer(dataset_train_all)
validation_init_op = iterator.make_initializer(dataset_dev_all)
with tf.name_scope('inputs'):
next_batch = iterator.get_next(name='next_batch')
names = []
detected = []
arousal = []
valence = []
liking = []
istalkings = []
images = []
# len(next_batch) is 34.
# len(n) is 7. Since we are extracting: name, detected, arousal, valence, liking, istalking and images...
# len(n[0 or 1 or 2 or ... or 6]) = is batch size.
for n in next_batch:
names.append(n[0])
detected.append(n[1])
arousal.append(n[2])
valence.append(n[3])
liking.append(n[4])
istalkings.append(n[5])
images.append(n[6])
names = tf.concat(names, axis=0, name='names')
detected = tf.concat(detected, axis=0, name='detected')
arousal = tf.concat(arousal, axis=0, name='arousal')
valence = tf.concat(valence, axis=0, name='valence')
liking = tf.concat(liking, axis=0, name='liking')
istalkings = tf.concat(istalkings, axis=0, name='istalkings')
images = tf.concat(images, axis=0, name='images')
return names, detected, arousal, valence, liking, istalkings, images, training_init_op, validation_init_op
Now if I try the following:
sess = tf.Session()
sess.run(training_init_op)
print(sess.run(names))
I got the following error:
ValueError: The two structures don't have the same number of elements.
which makes sense because the number of training files is 34 while that for validation dataset is 14.
I would like to know how can I achieve the goal in mind?
Any help is much appreciated!!