我是spark的新手,我正在尝试在一个伪分布式hadoop系统上运行scala作业。
hadoop2.6+yarn+spark1.6.1+scala2.10.6+jvm8,一切从头开始安装。
我的scala应用程序是一个简单的wordcount示例,我不知道是什么错误。
/usr/local/sparkapps/WordCount/src/main/scala/com/mydomain/spark/wordcount/WordCount.scala
package com.mydomain.spark.wordcount
import org.apache.spark.{SparkConf, SparkContext}
import org.apache.spark.SparkContext._
object ScalaWordCount {
def main(args: Array[String]) {
val logFile = "/home/hduser/inputfile.txt"
val sparkConf = new SparkConf().setAppName("Spark Word Count")
val sc = new SparkContext(sparkConf)
val file = sc.textFile(logFile)
val counts = file.flatMap(_.split(" ")).map(word => (word, 1)).reduceByKey(_ + _)
counts.saveAsTextFile("/home/hduser/output")
}
}
sbt文件:
/usr/local/sparkapps/WordCount/WordCount.sbt
name := "ScalaWordCount"
version := "1.0"
scalaVersion := "2.10.6"
libraryDependencies += "org.apache.spark" %% "spark-core" % "1.6.1"
编译:
$ cd /usr/local/sparkapps/WordCount/
$ sbt package
提交:
spark-submit --class com.mydomain.spark.wordcount.ScalaWordCount --master yarn-cluster /usr/local/sparkapps/WordCount/target/scala-2.10/scalawordcount_2.10-1.0.jar
输出:
Exception in thread "main" org.apache.spark.SparkException: Application application_1460107053907_0003 finished with failed status
at org.apache.spark.deploy.yarn.Client.run(Client.scala:1034)
at org.apache.spark.deploy.yarn.Client$.main(Client.scala:1081)
at org.apache.spark.deploy.yarn.Client.main(Client.scala)
at sun.reflect.NativeMethodAccessorImpl.invoke0(Native Method)
at sun.reflect.NativeMethodAccessorImpl.invoke(NativeMethodAccessorImpl.java:62)
at sun.reflect.DelegatingMethodAccessorImpl.invoke(DelegatingMethodAccessorImpl.java:43)
at java.lang.reflect.Method.invoke(Method.java:497)
at org.apache.spark.deploy.SparkSubmit$.org$apache$spark$deploy$SparkSubmit$$runMain(SparkSubmit.scala:731)
at org.apache.spark.deploy.SparkSubmit$.doRunMain$1(SparkSubmit.scala:181)
at org.apache.spark.deploy.SparkSubmit$.submit(SparkSubmit.scala:206)
at org.apache.spark.deploy.SparkSubmit$.main(SparkSubmit.scala:121)
at org.apache.spark.deploy.SparkSubmit.main(SparkSubmit.scala)
spark日志文件:http://pastebin.com/fnxfximm
1条答案
按热度按时间z18hc3ub1#
从日志:
如果要读取本地文件,请使用