Here's what I have:
gcloud dataproc workflow-templates create $TEMPLATE_ID --region $REGION
gcloud beta dataproc workflow-templates set-managed-cluster $TEMPLATE_ID --region $REGION --cluster-name dailyhourlygtp --image-version 1.5
--master-machine-type=n1-standard-8 --worker-machine-type=n1-standard-16 --num-workers=10 --master-boot-disk-size=500
--worker-boot-disk-size=500 --zone=europe-west1-b
export STEP_ID=step_pyspark1
gcloud dataproc workflow-templates add-job pyspark \
gs://$BUCKET_NAME/my_pyscript.py \
--step-id $STEP_ID \
--workflow-template $TEMPLATE_ID \
--region $REGION \
--jar=gs://spark-lib/bigquery/spark-bigquery-latest_2.12.jar
--initialization-actions gs://spark-lib/bigquery/spark-bigquery-latest_2.12.jar
--properties spark.jars.packages=gs://spark-lib/bigquery/spark-bigquery-latest_2.12.jar
gcloud dataproc workflow-templates instantiate $TEMPLATE_ID --region=$REGION
So here the question is how do I pass the following spark parameter to my my_pyscript.py:
--master yarn --deploy-mode cluster --conf "spark.sql.shuffle.partitions=900"
--conf "spark.sql.autoBroadcastJoinThreshold=10485760" --conf "spark.executor.memoryOverhead=8192"
--conf "spark.dynamicAllocation.enabled=true" --conf "spark.shuffle.service.enabled=true"
--executor-cores 5 --executor-memory 15g --driver-memory 16g