Created
January 30, 2018 16:44
-
-
Save ianmilligan1/e376792b1efc7c6b5a1bc50eaf5eb858 to your computer and use it in GitHub Desktop.
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| ubuntu@indexer:~/aut-plumbing/Scripts$ cat university-of-alberta-websites.log | |
| 2018-01-30 16:11:08,030 [main] WARN NativeCodeLoader - Unable to load native-hadoop library for your platform... using builtin-java classes where applicable | |
| 2018-01-30 16:11:08,106 [main] WARN Utils - Your hostname, indexer resolves to a loopback address: 127.0.0.1; using 192.168.32.19 instead (on interface ens3) | |
| 2018-01-30 16:11:08,106 [main] WARN Utils - Set SPARK_LOCAL_IP if you need to bind to another address | |
| 2018-01-30 16:11:13,858 [main] WARN ObjectStore - Failed to get database global_temp, returning NoSuchObjectException | |
| Spark context Web UI available at http://192.168.32.19:4040 | |
| Spark context available as 'sc' (master = local[12], app id = local-1517328668626). | |
| Spark session available as 'spark'. | |
| Loading university-of-alberta-websites.scala... | |
| import io.archivesunleashed.spark.matchbox.{ExtractDomain, ExtractLinks, RemoveHTML, RecordLoader, WriteGEXF} | |
| import io.archivesunleashed.spark.rdd.RecordRDD._ | |
| 2018-01-30 16:43:22,207 [dispatcher-event-loop-12] WARN HeartbeatReceiver - Removing executor driver with no recent heartbeats: 153931 ms exceeds timeout 120000 ms | |
| 2018-01-30 16:43:22,212 [dispatcher-event-loop-12] ERROR TaskSchedulerImpl - Lost executor driver on localhost: Executor heartbeat timed out after 153931 ms | |
| 2018-01-30 16:43:22,217 [driver-heartbeater] WARN NettyRpcEndpointRef - Error sending message [message = Heartbeat(driver,[Lscala.Tuple2;@79dd2f83,BlockManagerId(driver, 192.168.32.19, 41240, None))] in 1 attempts | |
| org.apache.spark.rpc.RpcTimeoutException: Futures timed out after [10 seconds]. This timeout is controlled by spark.executor.heartbeatInterval | |
| at org.apache.spark.rpc.RpcTimeout.org$apache$spark$rpc$RpcTimeout$$createRpcTimeoutException(RpcTimeout.scala:48) | |
| at org.apache.spark.rpc.RpcTimeout$$anonfun$addMessageIfTimeout$1.applyOrElse(RpcTimeout.scala:63) | |
| at org.apache.spark.rpc.RpcTimeout$$anonfun$addMessageIfTimeout$1.applyOrElse(RpcTimeout.scala:59) | |
| at scala.PartialFunction$OrElse.apply(PartialFunction.scala:167) | |
| at org.apache.spark.rpc.RpcTimeout.awaitResult(RpcTimeout.scala:83) | |
| at org.apache.spark.rpc.RpcEndpointRef.askWithRetry(RpcEndpointRef.scala:102) | |
| at org.apache.spark.executor.Executor.org$apache$spark$executor$Executor$$reportHeartBeat(Executor.scala:689) | |
| at org.apache.spark.executor.Executor$$anon$1$$anonfun$run$1.apply$mcV$sp(Executor.scala:718) | |
| at org.apache.spark.executor.Executor$$anon$1$$anonfun$run$1.apply(Executor.scala:718) | |
| at org.apache.spark.executor.Executor$$anon$1$$anonfun$run$1.apply(Executor.scala:718) | |
| at org.apache.spark.util.Utils$.logUncaughtExceptions(Utils.scala:1951) | |
| at org.apache.spark.executor.Executor$$anon$1.run(Executor.scala:718) | |
| at java.util.concurrent.Executors$RunnableAdapter.call(Executors.java:511) | |
| at java.util.concurrent.FutureTask.runAndReset(FutureTask.java:308) | |
| at java.util.concurrent.ScheduledThreadPoolExecutor$ScheduledFutureTask.access$301(ScheduledThreadPoolExecutor.java:180) | |
| at java.util.concurrent.ScheduledThreadPoolExecutor$ScheduledFutureTask.run(ScheduledThreadPoolExecutor.java:294) | |
| at java.util.concurrent.ThreadPoolExecutor.runWorker(ThreadPoolExecutor.java:1149) | |
| at java.util.concurrent.ThreadPoolExecutor$Worker.run(ThreadPoolExecutor.java:624) | |
| at java.lang.Thread.run(Thread.java:748) | |
| Caused by: java.util.concurrent.TimeoutException: Futures timed out after [10 seconds] | |
| at scala.concurrent.impl.Promise$DefaultPromise.ready(Promise.scala:219) | |
| at scala.concurrent.impl.Promise$DefaultPromise.result(Promise.scala:223) | |
| at scala.concurrent.Await$$anonfun$result$1.apply(package.scala:190) | |
| at scala.concurrent.BlockContext$DefaultBlockContext$.blockOn(BlockContext.scala:53) | |
| at scala.concurrent.Await$.result(package.scala:190) | |
| at org.apache.spark.rpc.RpcTimeout.awaitResult(RpcTimeout.scala:81) | |
| ... 14 more | |
| 2018-01-30 16:43:22,232 [dispatcher-event-loop-12] WARN TaskSetManager - Lost task 433.0 in stage 0.0 (TID 433, localhost, executor driver): ExecutorLostFailure (executor driver exited caused by one of the running tasks) Reason: Executor heartbeat timed out after 153931 ms | |
| 2018-01-30 16:43:22,243 [dispatcher-event-loop-12] ERROR TaskSetManager - Task 433 in stage 0.0 failed 1 times; aborting job | |
| 2018-01-30 16:43:22,244 [dispatcher-event-loop-12] WARN TaskSetManager - Lost task 424.0 in stage 0.0 (TID 424, localhost, executor driver): ExecutorLostFailure (executor driver exited caused by one of the running tasks) Reason: Executor heartbeat timed out after 153931 ms | |
| 2018-01-30 16:43:22,244 [dispatcher-event-loop-12] WARN TaskSetManager - Lost task 418.0 in stage 0.0 (TID 418, localhost, executor driver): ExecutorLostFailure (executor driver exited caused by one of the running tasks) Reason: Executor heartbeat timed out after 153931 ms | |
| 2018-01-30 16:43:22,244 [dispatcher-event-loop-12] WARN TaskSetManager - Lost task 430.0 in stage 0.0 (TID 430, localhost, executor driver): ExecutorLostFailure (executor driver exited caused by one of the running tasks) Reason: Executor heartbeat timed out after 153931 ms | |
| 2018-01-30 16:43:22,245 [dispatcher-event-loop-12] WARN TaskSetManager - Lost task 337.0 in stage 0.0 (TID 337, localhost, executor driver): ExecutorLostFailure (executor driver exited caused by one of the running tasks) Reason: Executor heartbeat timed out after 153931 ms | |
| 2018-01-30 16:43:22,245 [dispatcher-event-loop-12] WARN TaskSetManager - Lost task 420.0 in stage 0.0 (TID 420, localhost, executor driver): ExecutorLostFailure (executor driver exited caused by one of the running tasks) Reason: Executor heartbeat timed out after 153931 ms | |
| 2018-01-30 16:43:22,245 [dispatcher-event-loop-12] WARN TaskSetManager - Lost task 429.0 in stage 0.0 (TID 429, localhost, executor driver): ExecutorLostFailure (executor driver exited caused by one of the running tasks) Reason: Executor heartbeat timed out after 153931 ms | |
| 2018-01-30 16:43:22,245 [dispatcher-event-loop-12] WARN TaskSetManager - Lost task 432.0 in stage 0.0 (TID 432, localhost, executor driver): ExecutorLostFailure (executor driver exited caused by one of the running tasks) Reason: Executor heartbeat timed out after 153931 ms | |
| 2018-01-30 16:43:22,245 [dispatcher-event-loop-12] WARN TaskSetManager - Lost task 423.0 in stage 0.0 (TID 423, localhost, executor driver): ExecutorLostFailure (executor driver exited caused by one of the running tasks) Reason: Executor heartbeat timed out after 153931 ms | |
| 2018-01-30 16:43:22,245 [dispatcher-event-loop-12] WARN TaskSetManager - Lost task 426.0 in stage 0.0 (TID 426, localhost, executor driver): ExecutorLostFailure (executor driver exited caused by one of the running tasks) Reason: Executor heartbeat timed out after 153931 ms | |
| 2018-01-30 16:43:22,245 [dispatcher-event-loop-12] WARN TaskSetManager - Lost task 393.0 in stage 0.0 (TID 393, localhost, executor driver): ExecutorLostFailure (executor driver exited caused by one of the running tasks) Reason: Executor heartbeat timed out after 153931 ms | |
| 2018-01-30 16:43:22,245 [dispatcher-event-loop-12] WARN TaskSetManager - Lost task 431.0 in stage 0.0 (TID 431, localhost, executor driver): ExecutorLostFailure (executor driver exited caused by one of the running tasks) Reason: Executor heartbeat timed out after 153931 ms | |
| 2018-01-30 16:43:22,259 [dispatcher-event-loop-12] WARN NettyRpcEnv - Ignored message: HeartbeatResponse(true) | |
| 2018-01-30 16:43:22,262 [kill-executor-thread] WARN SparkContext - Killing executors is only supported in coarse-grained mode | |
| org.apache.spark.SparkException: Job aborted due to stage failure: Task 433 in stage 0.0 failed 1 times, most recent failure: Lost task 433.0 in stage 0.0 (TID 433, localhost, executor driver): ExecutorLostFailure (executor driver exited caused by one of the running tasks) Reason: Executor heartbeat timed out after 153931 ms | |
| Driver stacktrace: | |
| at org.apache.spark.scheduler.DAGScheduler.org$apache$spark$scheduler$DAGScheduler$$failJobAndIndependentStages(DAGScheduler.scala:1435) | |
| at org.apache.spark.scheduler.DAGScheduler$$anonfun$abortStage$1.apply(DAGScheduler.scala:1423) | |
| at org.apache.spark.scheduler.DAGScheduler$$anonfun$abortStage$1.apply(DAGScheduler.scala:1422) | |
| at scala.collection.mutable.ResizableArray$class.foreach(ResizableArray.scala:59) | |
| at scala.collection.mutable.ArrayBuffer.foreach(ArrayBuffer.scala:48) | |
| at org.apache.spark.scheduler.DAGScheduler.abortStage(DAGScheduler.scala:1422) | |
| at org.apache.spark.scheduler.DAGScheduler$$anonfun$handleTaskSetFailed$1.apply(DAGScheduler.scala:802) | |
| at org.apache.spark.scheduler.DAGScheduler$$anonfun$handleTaskSetFailed$1.apply(DAGScheduler.scala:802) | |
| at scala.Option.foreach(Option.scala:257) | |
| at org.apache.spark.scheduler.DAGScheduler.handleTaskSetFailed(DAGScheduler.scala:802) | |
| at org.apache.spark.scheduler.DAGSchedulerEventProcessLoop.doOnReceive(DAGScheduler.scala:1650) | |
| at org.apache.spark.scheduler.DAGSchedulerEventProcessLoop.onReceive(DAGScheduler.scala:1605) | |
| at org.apache.spark.scheduler.DAGSchedulerEventProcessLoop.onReceive(DAGScheduler.scala:1594) | |
| at org.apache.spark.util.EventLoop$$anon$1.run(EventLoop.scala:48) | |
| at org.apache.spark.scheduler.DAGScheduler.runJob(DAGScheduler.scala:628) | |
| at org.apache.spark.SparkContext.runJob(SparkContext.scala:1925) | |
| at org.apache.spark.SparkContext.runJob(SparkContext.scala:1938) | |
| at org.apache.spark.SparkContext.runJob(SparkContext.scala:1951) | |
| at org.apache.spark.SparkContext.runJob(SparkContext.scala:1965) | |
| at org.apache.spark.rdd.RDD$$anonfun$collect$1.apply(RDD.scala:936) | |
| at org.apache.spark.rdd.RDDOperationScope$.withScope(RDDOperationScope.scala:151) | |
| at org.apache.spark.rdd.RDDOperationScope$.withScope(RDDOperationScope.scala:112) | |
| at org.apache.spark.rdd.RDD.withScope(RDD.scala:362) | |
| at org.apache.spark.rdd.RDD.collect(RDD.scala:935) | |
| at org.apache.spark.RangePartitioner$.sketch(Partitioner.scala:266) | |
| at org.apache.spark.RangePartitioner.<init>(Partitioner.scala:128) | |
| at org.apache.spark.rdd.OrderedRDDFunctions$$anonfun$sortByKey$1.apply(OrderedRDDFunctions.scala:62) | |
| at org.apache.spark.rdd.OrderedRDDFunctions$$anonfun$sortByKey$1.apply(OrderedRDDFunctions.scala:61) | |
| at org.apache.spark.rdd.RDDOperationScope$.withScope(RDDOperationScope.scala:151) | |
| at org.apache.spark.rdd.RDDOperationScope$.withScope(RDDOperationScope.scala:112) | |
| at org.apache.spark.rdd.RDD.withScope(RDD.scala:362) | |
| at org.apache.spark.rdd.OrderedRDDFunctions.sortByKey(OrderedRDDFunctions.scala:61) | |
| at org.apache.spark.rdd.RDD$$anonfun$sortBy$1.apply(RDD.scala:619) | |
| at org.apache.spark.rdd.RDD$$anonfun$sortBy$1.apply(RDD.scala:620) | |
| at org.apache.spark.rdd.RDDOperationScope$.withScope(RDDOperationScope.scala:151) | |
| at org.apache.spark.rdd.RDDOperationScope$.withScope(RDDOperationScope.scala:112) | |
| at org.apache.spark.rdd.RDD.withScope(RDD.scala:362) | |
| at org.apache.spark.rdd.RDD.sortBy(RDD.scala:617) | |
| at io.archivesunleashed.spark.rdd.RecordRDD$CountableRDD.countItems(RecordRDD.scala:40) | |
| ... 73 elided | |
| 2018-01-30 16:43:25,149 [dispatcher-event-loop-10] ERROR TaskSchedulerImpl - Ignoring update with state FINISHED for TID 429 because its task set is gone (this is likely the result of receiving duplicate task finished status updates) or its executor has been marked as failed. | |
| 2018-01-30 16:43:27,309 [dispatcher-event-loop-2] ERROR TaskSchedulerImpl - Ignoring update with state FINISHED for TID 433 because its task set is gone (this is likely the result of receiving duplicate task finished status updates) or its executor has been marked as failed. | |
| 2018-01-30 16:43:34,153 [dispatcher-event-loop-14] ERROR TaskSchedulerImpl - Ignoring update with state FINISHED for TID 418 because its task set is gone (this is likely the result of receiving duplicate task finished status updates) or its executor has been marked as failed. | |
| ubuntu@indexer:~/aut-plumbing/Scripts$ |
Sign up for free
to join this conversation on GitHub.
Already have an account?
Sign in to comment