I am using 5D input_data [Batch, Depth, Height, Width, Channel]. I want to do a convolution only along the Width dimension and get an output the same size of the input.
I understood that tf.nn.depthwise_conv2d do that with 4D tensor input [Batch, Height, Width, Channel] but it does not accept 5D input.
I did this method that basically unstack 5D data relatively to the second dimension to get 4d tensor and then stack the results to get back 5D tensor:
import numpy as np
import matplotlib.pyplot as plt
import tensorflow as tf
import os
import time
def custom_depthwise_conv2d(data, kernel, strides, padding):
size_depth_input = data.shape[1]
if size_depth_input==1:
new_strides = list(strides)
new_strides.remove(1)
new_data = tf.squeeze(data, axis=1)
new_kernel = tf.squeeze(kernel, axis=0)
y = tf.nn.depthwise_conv2d(new_data, new_kernel, strides=new_strides, padding=padding)
output = tf.expand_dims(y, axis=1)
return output
else:
new_strides = list(strides)
new_strides.remove(1)
list_output_convolution_depth = []
new_kernel = tf.squeeze(kernel, axis=0)
for id_depth in range(size_depth_input):
depth_data = data[:,id_depth,:,:,:]
y = tf.nn.depthwise_conv2d(depth_data, new_kernel, strides=new_strides, padding=padding)
output = tf.expand_dims(y, axis=1)
list_output_convolution_depth.append(output)
output = tf.concat(list_output_convolution_depth, axis=1)
return output
input_data = tf.ones(shape=(60, 4, 4, 1000, 3), dtype = "float32")
kernel = tf.ones(shape=[1,1,10, 3, 1], dtype = "float32")
output = custom_depthwise_conv2d(input_data, kernel, strides=[1,1,1,1,1], padding="SAME")
print("input_data.shape", input_data.shape)
print("kernel.shape", kernel.shape)
print("output.shape", output.shape)
Is there a solution to do it more efficiently/elegantly ?