[3]<stderr>:Error in sys.excepthook:
[3]<stderr>:
[3]<stderr>:Original exception was:
[3]<stdout>:#
[3]<stdout>:# A fatal error has been detected by the Java Runtime Environment:
[3]<stdout>:#
[3]<stdout>:# SIGSEGV (0xb) at pc=0x000000000055d91b, pid=1592523, tid=1592523
[3]<stdout>:#
[3]<stdout>:# JRE version: OpenJDK Runtime Environment (11.0.4+11) (build 11.0.4+11)
[3]<stdout>:# Java VM: OpenJDK 64-Bit Server VM (11.0.4+11, mixed mode, tiered, compressed oops, g1 gc, linux-amd64)
[3]<stdout>:# Problematic frame:
[3]<stdout>:# C [python3.10+0x15d91b] PyObject_GC_UnTrack+0x1b
[3]<stdout>:#
[3]<stdout>:# Core dump will be written. Default location: Core dumps may be processed with "/usr/lib/systemd/systemd-coredump %P %u %g %s %t %c %h %e" (or dumping to /dfs/12/yarn/nm/usercache/userid/appcache/application_1718265469723_0399/container_e733_1718265469723_0399_01_000002/core.1592523)
[3]<stdout>:#
[3]<stdout>:# An error report file with more information is saved as:
[3]<stdout>:# /dfs/12/yarn/nm/usercache/userid/appcache/application_1718265469723_0399/container_e733_1718265469723_0399_01_000002/hs_err_pid1592523.log
[3]<stdout>:#
[3]<stdout>:# If you would like to submit a bug report, please visit:
[3]<stdout>:# http://bugreport.java.com/bugreport/crash.jsp
[3]<stdout>:# The crash happened outside the Java Virtual Machine in native code.
[3]<stdout>:# See problematic frame for where to report the bug.
[3]<stdout>:#
from tensorflow import keras
import horovod.spark.keras as hvd
from pyspark.sql import SparkSession
from pyspark.ml.feature import VectorAssembler
from horovod.spark.common.store import HDFSStore
import os
spark = SparkSession.builder. \
appName("HorovodTest"). \
config("spark.dynamicAllocation.enabled", False). \
config('spark.yarn.queue', 'queue_name'). \
config('spark.executor.instances', '4').\
config('spark.executor.cores', '8').\
config('spark.executor.memory', '20g').\
config('spark.driver.memory', '40g').\
config('spark.shuffle.io.maxRetries', '10').\
config('spark.shuffle.io.retryWait', '360s').\
config('spark.executor.memoryOverhead', '10g').\
config('spark.yarn.appMasterEnv.HOROVOD_GLOO_TIMEOUT_SECONDS', '3600').\
config('spark.executorEnv.HOROVOD_GLOO_TIMEOUT_SECONDS', '3600').\
config('spark.network.timeout', '18000s').master("yarn").getOrCreate()
file_path = "/user/userid/hvd_data/" # hdfs path
df = spark.read.parquet(file_path)
df = df.repartition(4)
feature_columns = df.columns[:-1] # all columns except the last one
label_column = df.columns[-1] # the last column
# Assemble the features into a single vector column
assembler = VectorAssembler(inputCols=feature_columns, outputCol="features")
df = assembler.transform(df)
# Define the Keras model
model = keras.Sequential([
keras.layers.Dense(8, input_dim=len(feature_columns)),
keras.layers.Activation('tanh'),
keras.layers.Dense(1),
keras.layers.Activation('sigmoid')
])
# Unscaled learning rate
optimizer = keras.optimizers.SGD(learning_rate=0.1)
loss = 'binary_crossentropy'
# Define the HDFS store for checkpointing
store = HDFSStore('/user/userid/tmp')
# Create Horovod Keras Estimator
keras_estimator = hvd.KerasEstimator(
num_proc=4,
store=store,
model=model,
optimizer=optimizer,
loss=loss,
feature_cols=['features'],
label_cols=['Fraud'],
batch_size=32,
epochs=4,
partitions_per_process=1
)
# Fit the model
keras_model = keras_estimator.fit(df)
print(">>>>>>>>>> Model Training Completed Successfully <<<<<<<<<<<<<<<<<<<<")
print(keras_model)
Note: It works fine if I run with num_proc=1 but running not this error when num_proc>1
Task Description:
Training a simple classifier using keras + horovod spark and getting below error
Error:
Environment:
Code Snippet:
Note: It works fine if I run with num_proc=1 but running not this error when num_proc>1