> I am facing this issue intermittently in dataproc spark job in GCP for both long running job for 3-4 hrs and short running job 25-20 min. below is the spark configuration and dags conf of GCP cluster `
spark.cores.max="80"
master="yarn"
spark.executor.memory="8G"
spark.task.maxFailures="50"
spark.driver.maxResultSize="10g"
spark.hadoop.dfs.replication= "1"
MASTER_MACHINE_TYPE = "n1-standard-4"
WORKER_MACHINE_TYPE = "n1-standard-16"
# Dataproc cluster definition
CLUSTER_CONFIG = {
"gce_cluster_config": {
"subnetwork_uri": SUBNETWORK_URI,
"internal_ip_only": True,
"service_account" : GCP_SERVICE_ACCOUNT
},
"master_config": {
"num_instances": 3,
"machine_type_uri": MASTER_MACHINE_TYPE,
"disk_config": {"boot_disk_type": "pd-standard", "boot_disk_size_gb": 1024},
"image_uri" : IMAGE_URI
},
"worker_config": {
"num_instances": 10,
"machine_type_uri": WORKER_MACHINE_TYPE,
"disk_config": {"boot_disk_type": "pd-standard", "boot_disk_size_gb": 1024},
"image_uri" : IMAGE_URI
},
"software_config": {
"properties": {
"dataproc:dataproc.allow.zero.workers": "true",
"yarn:yarn.resourcemanager.scheduler.class":
"org.apache.hadoop.yarn.server.resourcemanager.scheduler.fair.FairScheduler",
"yarn:yarn.nodemanager.resource.cpu-vcores": "48"
}
},
"endpoint_config" : {
"enable_http_port_access" : True
},
"lifecycle_config": {
"idle_delete_ttl": {"seconds": 15*60},
"auto_delete_ttl": {"seconds": 480*60},
}
}
`