I have the following script which works well for writing smaller datasets out to a file, but eventually runs out of memory when processing and writing larger datasets. Some files sizes will be 60gb +.
def do_work(index):
ref_feature = layer.GetFeature(index)
if ref_feature:
try:
return ref_feature.ExportToJson(as_object=True)
except Exception as e:
pass
return None
def run_mp():
# empty file contents
open(f"{out_dir}/{fc_name}.geojsonseq", "w", encoding='utf8').close()
# initiate multiprocessing
pool = Pool(cpu_count())
fc = layer.GetFeatureCount()
resultset = pool.imap_unordered(do_work, range(fc), chunksize=1000)
# this part is done after all results are ready, resulting in huge memory storage until results are written
with open(f"{out_dir}/{fc_name}.geojsonseq", 'a') as file:
for obj in resultset:
file.write(f"\x1e{json.dumps(obj)}\n")
if __name__ == '__main__':
seg_start = time.time()
run_mp()
print(f' completed in {time.time() - seg_start}')
Question:
Is there a way to stream the results directly out to a file without building it up in memory and dumping it out to a file at the end?