spark-shell批处理

#!/bin/bash

source /etc/profile

exec $SPARK_HOME/bin/spark-shell --queue tv --name spark-sql-test --executor-cores 8 --executor-memory 8g --num-executors 8 --conf spark.cleaner.ttl=240000 <

import org.apache.spark.sql.SaveMode

sql("set hive.exec.dynamic.partition=true")

sql("set hive.exec.dynamic.partition.mode=nonstrict")

sql("use hr")

sql("SELECT * FROM t_abc ").rdd.saveAsTextFile("/tmp/out")

sql("SELECT * FROM t_abc").rdd.map(_.toString).intersection(sc.textFile("/user/hdfs/t2_abc").map(_.toString).distinct).count

!EOF

你可能感兴趣的:(linux,spark)