Dask IndexError: list index out of range

Viewed 141

So i have folder called "data" say containing many CSV files

import dask.dataframe as dd
df = dd.read_csv('data/*.csv')
df.head()
df.column_1.mean().compute()

The above lines of code work perfectly and the dask.compute method does its job. But when i add the "include_path_column=True" parameter to the dd.read_cs() function call, i get the following error:

IndexError: list index out of range

When i expand the error i get

IndexError                                Traceback (most recent call last)
<ipython-input-129-4be67235bebb> in <module>
----> 1 df['H_hp'].mean().compute()

/anaconda/envs/azureml_py36/lib/python3.6/site-packages/dask/base.py in compute(self, **kwargs)
    165         dask.base.compute
    166         """
--> 167         (result,) = compute(self, traverse=False, **kwargs)
    168         return result
    169 

/anaconda/envs/azureml_py36/lib/python3.6/site-packages/dask/base.py in compute(*args, **kwargs)
    444     )
    445 
--> 446     dsk = collections_to_dsk(collections, optimize_graph, **kwargs)
    447     keys, postcomputes = [], []
    448     for x in collections:

/anaconda/envs/azureml_py36/lib/python3.6/site-packages/dask/base.py in collections_to_dsk(collections, optimize_graph, **kwargs)
    216             dsk, keys = _extract_graph_and_keys(val)
    217             groups[opt] = (dsk, keys)
--> 218             _opt = opt(dsk, keys, **kwargs)
    219             _opt_list.append(_opt)
    220 

/anaconda/envs/azureml_py36/lib/python3.6/site-packages/dask/dataframe/optimize.py in optimize(dsk, keys, **kwargs)
     19         dsk = fuse_roots(dsk, keys=flat_keys)
     20 
---> 21     dsk = ensure_dict(dsk)
     22 
     23     if isinstance(keys, list):

/anaconda/envs/azureml_py36/lib/python3.6/site-packages/dask/utils.py in ensure_dict(d)
   1030             dd_id = id(dd)
   1031             if dd_id not in seen:
-> 1032                 result.update(dd)
   1033                 seen.add(dd_id)
   1034         return result

/anaconda/envs/azureml_py36/lib/python3.6/site-packages/dask/dataframe/io/csv.py in __getitem__(self, key)
     80 
     81         if self.paths is not None:
---> 82             path_info = (self.colname, self.paths[i], self.paths)
     83         else:
     84             path_info = None

IndexError: list index out of range

0 Answers
Related