AzureML DatasetValidationError when using register_pandas_dataframe

Viewed 208

I want to register a pandas dataframe in azureml via the register_pandas_dataframe method. I struggle with it with quite some time now as simple googling does not yield any help (for azureml there is not much help, and many deprecated docs) and going through the error message i cant find i fix.

In this sample i replaced my dataframe with a very simple one for testing purposes.

df_temp = pd.DataFrame({'A' : [1,2,3], 'B' : [4,5,6]})
Dataset.Tabular.register_pandas_dataframe(df_temp,
                                          DATASTORE,
                                          "Test_dataset",
                                          description='Test',
                                          tags=None,
                                          show_progress=True)

Instead of registering it i get:


DataflowValidationError                   Traceback (most recent call last)
File /anaconda/envs/azureml_py38/lib/python3.8/site-packages/azureml/data/dataset_error_handling.py:65, in _validate_has_data(dataflow, error_message)
     64 try:
---> 65     dataflow.verify_has_data()
     66 except (dataprep().api.dataflow.DataflowValidationError,
     67         dataprep().api.errorhandlers.ExecutionError) as e:

File /anaconda/envs/azureml_py38/lib/python3.8/site-packages/azureml/dataprep/api/_loggerfactory.py:213, in track.<locals>.monitor.<locals>.wrapper(*args, **kwargs)
    212 try:
--> 213     return func(*args, **kwargs)
    214 except Exception as e:

File /anaconda/envs/azureml_py38/lib/python3.8/site-packages/azureml/dataprep/api/dataflow.py:835, in Dataflow.verify_has_data(self)
    834 if len(self.take(1)._to_pyrecords()) == 0:
--> 835     raise DataflowValidationError("The Dataflow produced no records.")

DataflowValidationError: 
Error Code: Empty
Error Message: The Dataflow produced no records.| session_id=41857052-3cf1-4cf8-93ac-fe7ab1fd94e7

During handling of the above exception, another exception occurred:

DatasetValidationError                    Traceback (most recent call last)
Input In [11], in <cell line: 12>()
      8 df_temp.to_csv('Processed Data/Bahn_2022.csv',index=False)
     10 df_temp = pd.DataFrame({'A' : [1,2,3], 'B' : [4,5,6]})
---> 12 Dataset.Tabular.register_pandas_dataframe(df_temp,
     13                                           DATASTORE,
     14                                           "Bahn_dataset",
     15                                           description='Datensatz aller aufgezeichneten Fahrten der Bahnen mit zusätzlichen Features.',
     16                                           tags=None,
     17                                           show_progress=True)

File /anaconda/envs/azureml_py38/lib/python3.8/site-packages/azureml/data/_loggerfactory.py:132, in track.<locals>.monitor.<locals>.wrapper(*args, **kwargs)
    130 with _LoggerFactory.track_activity(logger, func.__name__, activity_type, custom_dimensions) as al:
    131     try:
--> 132         return func(*args, **kwargs)
    133     except Exception as e:
    134         if hasattr(al, 'activity_info') and hasattr(e, 'error_code'):

File /anaconda/envs/azureml_py38/lib/python3.8/site-packages/azureml/data/dataset_factory.py:660, in TabularDatasetFactory.register_pandas_dataframe(dataframe, target, name, description, tags, show_progress)
    658 console("Creating and registering a new dataset.")
    659 datapath = DataPath(datastore, relative_path_with_guid)
--> 660 saved_dataset = TabularDatasetFactory.from_parquet_files(datapath)
    661 registered_dataset = saved_dataset.register(datastore.workspace, name,
    662                                             description=description,
    663                                             tags=tags,
    664                                             create_new_version=True)
    665 console("Successfully created and registered a new dataset.")

File /anaconda/envs/azureml_py38/lib/python3.8/site-packages/azureml/data/_loggerfactory.py:132, in track.<locals>.monitor.<locals>.wrapper(*args, **kwargs)
    130 with _LoggerFactory.track_activity(logger, func.__name__, activity_type, custom_dimensions) as al:
    131     try:
--> 132         return func(*args, **kwargs)
    133     except Exception as e:
    134         if hasattr(al, 'activity_info') and hasattr(e, 'error_code'):

File /anaconda/envs/azureml_py38/lib/python3.8/site-packages/azureml/data/dataset_factory.py:142, in TabularDatasetFactory.from_parquet_files(path, validate, include_path, set_column_types, partition_format)
    140 path = _validate_and_normalize_path(path)
    141 _trace_dataset_creation(path)
--> 142 return TabularDatasetFactory._from_parquet_files(path,
    143                                                  validate,
    144                                                  include_path,
    145                                                  set_column_types,
    146                                                  partition_format)

File /anaconda/envs/azureml_py38/lib/python3.8/site-packages/azureml/data/dataset_factory.py:156, in TabularDatasetFactory._from_parquet_files(path, validate, include_path, set_column_types, partition_format)
    151 from azureml.data import TabularDataset
    153 dataflow = dataprep().read_parquet_file(path,
    154                                         include_path=True,
    155                                         verify_exists=False)
--> 156 dataflow = _transform_and_validate(
    157     dataflow, partition_format, include_path,
    158     validate or _is_inference_required(set_column_types))
    159 dataflow = _set_column_types(dataflow, set_column_types)
    160 return TabularDataset._create(dataflow)

File /anaconda/envs/azureml_py38/lib/python3.8/site-packages/azureml/data/dataset_factory.py:1166, in _transform_and_validate(dataflow, partition_format, include_path, validate, infer_column_types)
   1164     dataflow = dataflow.drop_columns('Path')
   1165 if validate:
-> 1166     _validate_has_data(dataflow, 'Failed to validate the data.')
   1167 elif infer_column_types:
   1168     _validate_has_data(dataflow, 'Failed to infer column type.'
   1169                                  'if data is inaccessible, please set infer_column_types to False.')

File /anaconda/envs/azureml_py38/lib/python3.8/site-packages/azureml/data/dataset_error_handling.py:68, in _validate_has_data(dataflow, error_message)
     65     dataflow.verify_has_data()
     66 except (dataprep().api.dataflow.DataflowValidationError,
     67         dataprep().api.errorhandlers.ExecutionError) as e:
---> 68     raise DatasetValidationError(error_message + '\n' + e.compliant_message, exception=e)

DatasetValidationError: DatasetValidationError:
    Message: Failed to validate the data.
The Dataflow produced no records.| session_id=41857052-3cf1-4cf8-93ac-fe7ab1fd94e7
    InnerException None
    ErrorResponse 
{
    "error": {
        "code": "UserError",
        "message": "Failed to validate the data.\nThe Dataflow produced no records.| session_id=41857052-3cf1-4cf8-93ac-fe7ab1fd94e7"
    }
}

Ideas? I've ran out of them.

0 Answers
Related