I have created a model using diabetes dataset for prediction. I have trained, evaluated, logged and registered it as a new model in ML flow. Now I am trying to load the registered model and trying to predict on new data. All though I was able to predict the results. I am not able to display it. When I try to display using command .show() or display() it is throwing an error. What is the cause of the error? and How do I display the results?
Note: I have programmed using pure pyspark and all the ML flow operation was done on Data bricks
Code:-
model_details = mlflow.tracking.MlflowClient().get_latest_versions('model1',stages=['staging'])[0]
model = mlflow.pyfunc.spark_udf(spark,model_details.source)
input_df = sdf.drop('progression')
columns = list(map(lambda c: f"{c}", input_df.columns))
df = input_df.withColumn("progression", model(*columns))
df.show(truncate=False)
Error :-
PythonException: An exception was thrown from a UDF: 'Exception: Java gateway process exited before sending its port number'. Full traceback below:
PythonException Traceback (most recent call last)
<command-1343735193245452> in <module>
34 df = input_df.withColumn("progression", model(*columns))
35
---> 36 df.show(truncate=False)
/databricks/spark/python/pyspark/sql/dataframe.py in show(self, n, truncate, vertical)
441 print(self._jdf.showString(n, 20, vertical))
442 else:
--> 443 print(self._jdf.showString(n, int(truncate), vertical))
444
445 def __repr__(self):
/databricks/spark/python/lib/py4j-0.10.9-src.zip/py4j/java_gateway.py in __call__(self, *args)
1303 answer = self.gateway_client.send_command(command)
1304 return_value = get_return_value(
-> 1305 answer, self.gateway_client, self.target_id, self.name)
1306
1307 for temp_arg in temp_args:
/databricks/spark/python/pyspark/sql/utils.py in deco(*a, **kw)
131 # Hide where the exception came from that shows a non-Pythonic
132 # JVM exception message.
--> 133 raise_from(converted)
134 else:
135 raise
/databricks/spark/python/pyspark/sql/utils.py in raise_from(e)
PythonException: An exception was thrown from a UDF: 'Exception: Java gateway process exited before sending its port number'. Full traceback below:
Traceback (most recent call last):
File "/databricks/spark/python/pyspark/worker.py", line 654, in main
process()
File "/databricks/spark/python/pyspark/worker.py", line 646, in process
serializer.dump_stream(out_iter, outfile)
File "/databricks/spark/python/pyspark/sql/pandas/serializers.py", line 281, in dump_stream
timely_flush_timeout_ms=self.timely_flush_timeout_ms)
File "/databricks/spark/python/pyspark/sql/pandas/serializers.py", line 97, in dump_stream
for batch in iterator:
File "/databricks/spark/python/pyspark/sql/pandas/serializers.py", line 271, in init_stream_yield_batches
for series in iterator:
File "/databricks/spark/python/pyspark/worker.py", line 467, in mapper
result = tuple(f(*[a[o] for o in arg_offsets]) for (arg_offsets, f) in udfs)
File "/databricks/spark/python/pyspark/worker.py", line 467, in <genexpr>
result = tuple(f(*[a[o] for o in arg_offsets]) for (arg_offsets, f) in udfs)
File "/databricks/spark/python/pyspark/worker.py", line 111, in <lambda>
verify_result_type(f(*a)), len(a[0])), arrow_return_type)
File "/databricks/spark/python/pyspark/util.py", line 109, in wrapper
return f(*args, **kwargs)
File "/databricks/python/lib/python3.7/site-packages/mlflow/pyfunc/__init__.py", line 827, in predict
model = SparkModelCache.get_or_load(archive_path)
File "/databricks/python/lib/python3.7/site-packages/mlflow/pyfunc/spark_model_cache.py", line 64, in get_or_load
SparkModelCache._models[archive_path] = load_pyfunc(temp_dir)
File "/databricks/python/lib/python3.7/site-packages/mlflow/utils/annotations.py", line 43, in deprecated_func
return func(*args, **kwargs)
File "/databricks/python/lib/python3.7/site-packages/mlflow/pyfunc/__init__.py", line 693, in load_pyfunc
return load_model(model_uri, suppress_warnings)
File "/databricks/python/lib/python3.7/site-packages/mlflow/pyfunc/__init__.py", line 667, in load_model
model_impl = importlib.import_module(conf[MAIN])._load_pyfunc(data_path)
File "/databricks/python/lib/python3.7/site-packages/mlflow/spark.py", line 707, in _load_pyfunc
.master("local[1]")
File "/databricks/spark/python/pyspark/sql/session.py", line 189, in getOrCreate
sc = SparkContext.getOrCreate(sparkConf)
File "/databricks/spark/python/pyspark/context.py", line 384, in getOrCreate
SparkContext(conf=conf or SparkConf())
File "/databricks/spark/python/pyspark/context.py", line 134, in __init__
SparkContext._ensure_initialized(self, gateway=gateway, conf=conf)
File "/databricks/spark/python/pyspark/context.py", line 333, in _ensure_initialized
SparkContext._gateway = gateway or launch_gateway(conf)
File "/databricks/spark/python/pyspark/java_gateway.py", line 105, in launch_gateway
raise Exception("Java gateway process exited before sending its port number")
Exception: Java gateway process exited before sending its port number