Same input, same model, same weights but getting different results

Viewed 275

I'm finetuning sentence-bert to do some task like sentence cosine-similarity calculation in Tensorflow. I set up a encoder, let's say, encoder1 using the code below:

from sentence_transformers import SentenceTransformer
sentences = ["This is an example sentence", "Each sentence is converted"]
model = SentenceTransformer('sentence-transformers/all-MiniLM-L12-v2')
embeddings = model.encode(sentences)

This is using the sentence-transformers API. And I also set up another encoder, call encoder2 using the code below:

from transformers import AutoTokenizer, TFAutoModel
import tensorflow as tf

tokenizer = AutoTokenizer.from_pretrained('sentence-transformers/all-MiniLM-L12-v2', from_pt=True)
model_tf = TFAutoModel.from_pretrained('sentence-transformers/all-MiniLM-L12-v2', from_pt=True)

encoded_input = tokenizer(sentences, padding=True, truncation=True, max_length=128, return_tensors='tf')
outputs = model_tf(**encoded_input)

def mean_pooling(model_output, input_mask):
    # seq_output shape=[batch_size, max_seq_len, hidden_size]
    # input_mask shape=[batch_size, max_seq_len]
    # expand input_mask
    seq_output = model_output[0]
    input_mask_expanded = tf.cast(tf.broadcast_to(tf.expand_dims(input_mask, -1), seq_output.shape), tf.float32)
    # pooled = tf.reduce_sum(seq_output * input_mask_expanded, 1) / tf.clip_by_value(tf.reduce_sum(input_mask_expanded, 1), clip_value_min=-1, clip_value_max=)
    pooled = tf.reduce_sum(seq_output * input_mask_expanded, 1) / tf.reduce_sum(input_mask_expanded, 1)
    # shape = [batch_size, hidden_size]
    return pooled

sentence_embeddings = mean_pooling(outputs, encoded_input['attention_mask'])

pooled_output, _ = tf.linalg.normalize(sentence_embeddings, 2, axis=1)

This loads a pre-trained model from Huggingface, which I have tested and I'm sure that it will produce the same results(pooled_output and embeddings) as in encoder1.

However, the weird thing is that, when I load this encoder2 into my tf.Model, and try to run a classifier to see whether two sentences are close, with the same input, same model trainable weights, same model, it gives different values. Does the network randomly initialize everything after I load the model?

Here's my encoder code:

class ApplicationCLS(tf.keras.layers.Layer):
    def __init__(self, bert_encoder_path, batch_size):
        super().__init__()
        self.bert_encoder = TFAutoModel.from_pretrained(bert_encoder_path, from_pt=True)
        self.classifier = CLSlayer(256, 1)
        self.loss_fn = tf.keras.losses.BinaryCrossentropy(from_logits=True)
        self.metric_fn = tf.keras.metrics.BinaryAccuracy(name="accuracy")
        self.auc_fn = tf.keras.metrics.AUC()
        self.batch_size = batch_size


    def mean_pooling(self, model_output, input_mask):
        seq_output = model_output[0]
        shape = [self.batch_size, seq_output.shape[1], seq_output.shape[2]]
        input_mask_expanded = tf.cast(tf.broadcast_to(tf.expand_dims(input_mask, -1), shape), tf.float32)
        pooled = tf.reduce_sum(seq_output * input_mask_expanded, 1) / tf.reduce_sum(input_mask_expanded, 1)
        return pooled


    def call(self, inputs, labels, training=True):
        sent1_inputs = inputs["sent1_inputs"]
        sent2_inputs = inputs["sent2_inputs"]
        # inputs: {"input_ids": input_ids, "input_mask":input_mask, "type_ids": type_id}
        sent1_outputs = self.bert_encoder(**sent1_inputs)
        tf.print("output:", sent1_outputs[0][0])
        sent1_embeddings = self.mean_pooling(sent1_outputs, sent1_inputs['attention_mask'])
        sent1_pooled_output, _ = tf.linalg.normalize(sent1_embeddings, 2, axis=1)

        sent2_outputs = self.bert_encoder(**sent2_inputs)
        sent2_embeddings = self.mean_pooling(sent2_outputs, sent2_inputs['attention_mask'])
        sent2_pooled_output, _ = tf.linalg.normalize(sent2_embeddings, 2, axis=1)

        # concat
        interaction = tf.concat([sent1_pooled_output, sent2_pooled_output], 1)

        # classification
        logits = self.classifier(interaction)

        loss = self.loss_fn(labels, logits)

        self.add_loss(loss)
        acc = self.metric_fn(labels, logits)
        auc = self.auc_fn(labels, logits)

        self.add_metric(loss, name="loss")
        self.add_metric(acc, name="acc")
        self.add_metric(auc, name="auc")

        return tf.nn.softmax(logits, name="prediction")

where the sent1_inputs is the same, but the printed outputs are different, what happened?

0 Answers
Related