I am working on the text summarization task and trying to add a .csv dataset to tensorflow_datasets (this is required to run a pre-trained transformer). I am following this tutorial https://www.tensorflow.org/datasets/add_dataset but I still don't get how to add it.
This is what I have so far:
import tensorflow_datasets.public_api as tfds
# TODO(data.csv): BibTeX citation
_CITATION = """
"""
_HOMEPAGE = "https:..."
# TODO(data.csv):
_DESCRIPTION = """A textual corpus of ...
"""
_DOCUMENT = "text"
_SUMMARY = "summary"
manual_dir = './'
class new_dataset(tfds.core.GeneratorBasedBuilder):
"""TODO(data.csv): Short description of my dataset."""
# TODO(data.csv): Set up version.
VERSION = tfds.core.Version('0.1.0')
def _info(self):
return tfds.core.DatasetInfo(
builder=self,
description=_DESCRIPTION,
features=tfds.features.FeaturesDict({
_DOCUMENT: tfds.features.Text(),
_SUMMARY: tfds.features.Text()
}),
supervised_keys=(_DOCUMENT, _SUMMARY),
homepage="https://...",
citation=_CITATION,
)
def _split_generators(self, dl_manager):
"""Returns SplitGenerators."""
# TODO(data.csv): Downloads the data and defines the splits
# dl_manager is a tfds.download.DownloadManager that can be used to
# download and extract URLs
return [
tfds.core.SplitGenerator(
name=tfds.Split.TRAIN,
# These kwargs will be passed to _generate_examples
gen_kwargs={},
),
]
def _generate_examples(self):
# Yields examples from the dataset
yield 'key', {}
How to define def _split_generators and def _generate_examples properly if my dataset is a .csv file with 2 columns: 'text' and 'summary'? This python file and my dataset are in the same directory.