I have nested for-loop in python to create a netCDF file. The for-loop takes a pandas dataframe with time, lat, lot, and parameters and replaces the information in the netCDF file by the parameters in the correct location and time. This is taking too long since the pandas dataframe has more than 80000 rows and the netCDF file has around 8000 time-steps. I've been looking to use either xargs or multiprocessing but in the first case the uses files as inputs and in the second case, it produces as many outputs as processes I use. I have no experience in parallel processing so probably my affirmations are totally wrong. This is the code that I am using:
with Dataset(os.path.join('Downloads', inv, 'observations.nc'), 'w') as dset:
dset.createDimension('time_components', 6)
groups = ['obs', 'mix_apri', 'mix_apos', 'mix_background']
for group in groups:
dset.createGroup(group)
dset[group].createDimension('nt', 8760)
dset[group].createDimension('nlat', 80)
dset[group].createDimension('nlon', 100)
times_start = dset[group].createVariable('times_start', 'i4', ('nt', 'time_components'))
times_end = dset[group].createVariable('times_end', 'i4', ('nt', 'time_components'))
lats = dset[group].createVariable('lats', 'f4', ('nlat'))
lons = dset[group].createVariable('lons', 'f4', ('nlon'))
times_start[:,:] = list(emis_apri['biosphere']['times_start'])
times_end[:,:] = list(emis_apri['biosphere']['times_end'])
lats[:] = list(emis_apri['biosphere']['lats'])
lons[:] = list(emis_apri['biosphere']['lons'])
conc_obs = dset['obs'].createVariable('conc', 'f8', ('nt', 'nlat', 'nlon'))
conc_mix_apri = dset['mix_apri'].createVariable('conc', 'f8', ('nt', 'nlat', 'nlon'))
conc_mix_apos = dset['mix_apos'].createVariable('conc', 'f8', ('nt', 'nlat', 'nlon'))
conc_mix_background = dset['mix_background'].createVariable('conc', 'f8', ('nt', 'nlat', 'nlon'))
for i in range(8760):
conc_obs[i,:,:] = emis_apri['biosphere']['emis'][i][:,:]*0
conc_mix_apri[:,:,:] = list(conc_obs)
conc_mix_apos[:,:,:] = list(conc_obs)
conc_mix_background[:,:,:] = list(conc_obs)
db = obsdb(os.path.join('Downloads', inv, 'observations.apos.tar.gz'))
nsites = db.sites.shape[0]
for isite, site in enumerate(db.sites.itertuples()):
dbs = db.observations.loc[db.observations.site == site.Index]
lat = where((array(emis_apri['biosphere']['lats']) >= list(dbs.lat)[0]-0.25) & (array(emis_apri['biosphere']['lats']) <= list(dbs.lat)[0]+0.25))[0][0]
lon = where((array(emis_apri['biosphere']['lons']) >= list(dbs.lon)[0]-0.25) & (array(emis_apri['biosphere']['lons']) <= list(dbs.lon)[0]+0.25))[0][0]
for i in range(len(list(dbs.time))):
for j in range(len(times_start)):
if datetime(*times_start[j,:].data) >= Timestamp.to_pydatetime(list(dbs.time)[i]) and datetime(*times_end[j,:].data) >= Timestamp.to_pydatetime(list(dbs.time)[i]):
conc_obs[i,lat,lon] = list(dbs.obs)[i]
conc_mix_apri[i,lat,lon] = list(dbs.mix_apri)[i]
conc_mix_apos[i,lat,lon] = list(dbs.mix_apos)[i]
conc_mix_background[i,lat,lon] = list(dbs.mix_background)[i]
From for isite, site in enumerate(db.sites.itertuples()): is the part of the code that I need to parallelize. I really appreciate any insights about this.