环境变量参考【CentOS6-大数据套件HA安装(1)统一环境配置)】
此处所有机器防火墙关闭,实际可根据需要调整。软件包统一在/usr/local/soft 目录,安装目录为:/hadoop
安装 spark
- 解压
tar -xzvf /usr/local/soft/spark-2.2.1-bin-hadoop2.7.tgz -C /hadoo
- 配置 spark-env.sh
cp spark-2.2.1-bin-hadoop2.7/conf/spark-env.sh.template spark-2.2.1-bin-hadoop2.7/conf/spark-env.sh
vi spark-2.2.1-bin-hadoop2.7/conf/spark-env.sh
export JAVA_HOME=/usr/java/jdk1.8.0_172-amd64
export SCALA_HOME=/usr/scala/scala-2.12.5
export HADOOP_HOME=/hadoop/hadoop-2.7.6
export HADOOP_CONF_DIR=$HADOOP_HOME/etc/hadoop
export SPARK_HOME=/hadoop/spark-2.2.1-bin-hadoop2.7
- 配置 spark在hdfs上的jar目录
# 在hdfs上创建目录
hdfs dfs -mkdir -p /user/spark/jars
# 将spark的jar上传到hdfs
hdfs dfs -put /hadoop/spark-2.2.1-bin-hadoop2.7/jars/* /user/spark/jars/
# 配置指向hdfs的目录
cp spark-2.2.1-bin-hadoop2.7/conf/spark-defaults.conf.template spark-2.2.1-bin-hadoop2.7/conf/spark-defaults.conf
vi spark-2.2.1-bin-hadoop2.7/conf/spark-defaults.conf
spark.yarn.jars hdfs://ns1/user/spark/jars/*
- 分发其他节点
scp -r spark-2.2.1-bin-hadoop2.7 hadoop@hadoop101:/hadoop/
scp -r spark-2.2.1-bin-hadoop2.7 hadoop@hadoop102:/hadoop/
scp -r spark-2.2.1-bin-hadoop2.7 hadoop@hadoop103:/hadoop/
scp -r spark-2.2.1-bin-hadoop2.7 hadoop@manager203:/hadoop/
- 测试
./spark-2.2.1-bin-hadoop2.7/bin/spark-submit --master yarn --deploy-mode cluster --class org.apache.spark.examples.SparkPi --executor-memory 512M --num-executors 1 /hadoop/spark-2.2.1-bin-hadoop2.7/examples/jars/spark-examples_2.11-2.2.1.jar 10