I am having difficulties understanding the tokenizer.pad method from the huggingface transformers library. In order to optimize training, I am performing tokenization in the Dataset such that no complicated operations are performed during data loading. My dataset looks like:
class DatasetTokenized(Dataset):
def __init__(self, data: pd.DataFrame, text_column: str,
label_columns: List[str], tokenizer_name: str):
super(DatasetTokenized, self).__init__()
self.data = data
self.text_column = text_column
self.label_columns = label_columns
self.tokenizer = BertTokenizer.from_pretrained(tokenizer_name)
self.tokenized_data = self.tokenize_data(data)
def __len__(self) -> int:
return len(self.tokenized_data)
def __getitem__(self, index: int) -> Dict:
return self.tokenized_data[index]
def tokenize_data(self, data: pd.DataFrame):
tokenized_data = []
print('Tokenizing data:')
for _, row in tqdm(data.iterrows(), total=len(data)):
text = row[self.text_column]
labels = row[self.label_columns]
encoding = self.tokenizer(text,
add_special_tokens=True,
max_length=512,
padding=False,
truncation=True,
return_attention_mask=True,
return_tensors='pt')
tokenized_data.append({
'text': text,
'encoding': encoding,
'labels': torch.FloatTensor(labels)
})
return tokenized_data
and my collator looks like:
class BertCollatorTokenized:
def __init__(self, tokenizer_name: str):
super(BertCollatorTokenized, self).__init__()
self.tokenizer = BertTokenizer.from_pretrained(tokenizer_name)
def __call__(self, batch: List[Any]):
text, encodings, labels = zip(
*[[sample['text'], sample['encoding'], sample['labels']]
for sample in batch])
encodings = list(encodings)
encodings = self.tokenizer.pad(encodings,
max_length=512,
padding='longest',
return_tensors='pt')
return {
'text': text,
'input_ids': encodings['input_ids'],
'attention_mask': encodings['attention_mask'],
'labels': torch.FloatTensor(labels)
}
I have confirmed that encodings is a list of BatchEncoding as required by tokenizer.pad. However, I am getting the following error:
ValueError: Unable to create tensor, you should probably activate truncation and/or padding with 'padding=True' 'truncation=True' to have batched tensors with the same length.
Which is a bit confusing as that should be the whole point of using tokenizer.pad. Anyone has an idea of what's going on?