Linux巩固记录（5） hadoop 2.7.4下自己编译代码并运行MapReduce程序

程序代码为 ~\hadoop-2.7.4\share\hadoop\mapreduce\sources\hadoop-mapreduce-examples-2.7.4-sources\org\apache\hadoop\examples\WordCount.java

第一次删除了package

import java.io.IOException;

import java.util.StringTokenizer;

import org.apache.hadoop.conf.Configuration;

import org.apache.hadoop.fs.Path;

import org.apache.hadoop.io.IntWritable;

import org.apache.hadoop.io.Text;

import org.apache.hadoop.mapreduce.Job;

import org.apache.hadoop.mapreduce.Mapper;

import org.apache.hadoop.mapreduce.Reducer;

import org.apache.hadoop.mapreduce.lib.input.FileInputFormat;

import org.apache.hadoop.mapreduce.lib.output.FileOutputFormat;

import org.apache.hadoop.util.GenericOptionsParser;

public class WordCount {

  public static class TokenizerMapper

       extends Mapper<Object, Text, Text, IntWritable>{

    private final static IntWritable one = new IntWritable(1);

    private Text word = new Text();

    public void map(Object key, Text value, Context context

                    ) throws IOException, InterruptedException {

      StringTokenizer itr = new StringTokenizer(value.toString());

      while (itr.hasMoreTokens()) {

        word.set(itr.nextToken());

        context.write(word, one);

      }

    }

  }

  public static class IntSumReducer

       extends Reducer<Text,IntWritable,Text,IntWritable> {

    private IntWritable result = new IntWritable();

    public void reduce(Text key, Iterable<IntWritable> values,

                       Context context

                       ) throws IOException, InterruptedException {

      int sum = 0;

      for (IntWritable val : values) {

        sum += val.get();

      }

      result.set(sum);

      context.write(key, result);

    }

  }

  public static void main(String[] args) throws Exception {

    Configuration conf = new Configuration();

    String[] otherArgs = new GenericOptionsParser(conf, args).getRemainingArgs();

    if (otherArgs.length < 2) {

      System.err.println("Usage: wordcount <in> [<in>...] <out>");

      System.exit(2);

    }

    Job job = Job.getInstance(conf, "word count");

    job.setJarByClass(WordCount.class);

    job.setMapperClass(TokenizerMapper.class);

    job.setCombinerClass(IntSumReducer.class);

    job.setReducerClass(IntSumReducer.class);

    job.setOutputKeyClass(Text.class);

    job.setOutputValueClass(IntWritable.class);

    for (int i = 0; i < otherArgs.length - 1; ++i) {

      FileInputFormat.addInputPath(job, new Path(otherArgs[i]));

    }

    FileOutputFormat.setOutputPath(job,

      new Path(otherArgs[otherArgs.length - 1]));

    System.exit(job.waitForCompletion(true) ? 0 : 1);

  }

}

此程序需要下面三个jar包才能编译通过

[root@master classes]# tree /home/jars/

/home/jars/

├── commons-cli-1.4.jar

├── hadoop-common-2.7.4.jar

└── hadoop-mapreduce-client-core-2.7.4.jar

执行过程及结果如下

[root@master classes]#

[root@master classes]# pwd

/home/classes

[root@master classes]# tree

.

0 directories, 0 files

[root@master classes]# tree /home/javaFile/

/home/javaFile/

└── WordCount.java

0 directories, 1 file

[root@master classes]# tree /home/jars/

/home/jars/

├── commons-cli-1.4.jar

├── hadoop-common-2.7.4.jar

└── hadoop-mapreduce-client-core-2.7.4.jar

0 directories, 3 files

[root@master classes]# javac -classpath .:/home/jars/* -d /home/classes/ /home/javaFile/WordCount.java

[root@master classes]# tree

.

├── WordCount.class

├── WordCount$IntSumReducer.class

└── WordCount$TokenizerMapper.class

0 directories, 3 files

[root@master classes]# jar -cvf wordc.jar ./*.class

added manifest

adding: WordCount.class(in = 1907) (out= 1040)(deflated 45%)

adding: WordCount$IntSumReducer.class(in = 1739) (out= 742)(deflated 57%)

adding: WordCount$TokenizerMapper.class(in = 1736) (out= 753)(deflated 56%)

[root@master classes]# tree

.

├── wordc.jar

├── WordCount.class

├── WordCount$IntSumReducer.class

└── WordCount$TokenizerMapper.class

0 directories, 4 files

[root@master classes]# /home/hadoop-2.7.4/bin/hadoop jar /home/classes/wordc.jar WordCount /hdfs-input.txt /result-self-compile

17/09/02 02:11:45 INFO client.RMProxy: Connecting to ResourceManager at master/192.168.0.80:8032

17/09/02 02:11:47 INFO input.FileInputFormat: Total input paths to process : 1

17/09/02 02:11:47 INFO mapreduce.JobSubmitter: number of splits:1

17/09/02 02:11:47 INFO mapreduce.JobSubmitter: Submitting tokens for job: job_1504320356950_0010

17/09/02 02:11:47 INFO impl.YarnClientImpl: Submitted application application_1504320356950_0010

17/09/02 02:11:47 INFO mapreduce.Job: The url to track the job: http://master:8088/proxy/application_1504320356950_0010/

17/09/02 02:11:47 INFO mapreduce.Job: Running job: job_1504320356950_0010

17/09/02 02:11:56 INFO mapreduce.Job: Job job_1504320356950_0010 running in uber mode : false

17/09/02 02:11:56 INFO mapreduce.Job:  map 0% reduce 0%

17/09/02 02:12:02 INFO mapreduce.Job:  map 100% reduce 0%

17/09/02 02:12:09 INFO mapreduce.Job:  map 100% reduce 100%

17/09/02 02:12:09 INFO mapreduce.Job: Job job_1504320356950_0010 completed successfully

17/09/02 02:12:10 INFO mapreduce.Job: Counters: 49

    File System Counters

        FILE: Number of bytes read=118

        FILE: Number of bytes written=241697

        FILE: Number of read operations=0

        FILE: Number of large read operations=0

        FILE: Number of write operations=0

        HDFS: Number of bytes read=174

        HDFS: Number of bytes written=76

        HDFS: Number of read operations=6

        HDFS: Number of large read operations=0

        HDFS: Number of write operations=2

    Job Counters

        Launched map tasks=1

        Launched reduce tasks=1

        Data-local map tasks=1

        Total time spent by all maps in occupied slots (ms)=3745

        Total time spent by all reduces in occupied slots (ms)=4081

        Total time spent by all map tasks (ms)=3745

        Total time spent by all reduce tasks (ms)=4081

        Total vcore-milliseconds taken by all map tasks=3745

        Total vcore-milliseconds taken by all reduce tasks=4081

        Total megabyte-milliseconds taken by all map tasks=3834880

        Total megabyte-milliseconds taken by all reduce tasks=4178944

    Map-Reduce Framework

        Map input records=6

        Map output records=12

        Map output bytes=118

        Map output materialized bytes=118

        Input split bytes=98

        Combine input records=12

        Combine output records=9

        Reduce input groups=9

        Reduce shuffle bytes=118

        Reduce input records=9

        Reduce output records=9

        Spilled Records=18

        Shuffled Maps =1

        Failed Shuffles=0

        Merged Map outputs=1

        GC time elapsed (ms)=155

        CPU time spent (ms)=1430

        Physical memory (bytes) snapshot=299466752

        Virtual memory (bytes) snapshot=4159479808

        Total committed heap usage (bytes)=141385728

    Shuffle Errors

        BAD_ID=0

        CONNECTION=0

        IO_ERROR=0

        WRONG_LENGTH=0

        WRONG_MAP=0

        WRONG_REDUCE=0

    File Input Format Counters

        Bytes Read=76

    File Output Format Counters

        Bytes Written=76

[root@master classes]# /home/hadoop-2.7.4/bin/hadoop fs -ls /

Found 3 items

-rw-r--r--   2 root supergroup         76 2017-09-02 00:57 /hdfs-input.txt

drwxr-xr-x   - root supergroup          0 2017-09-02 02:12 /result-self-compile

drwx------   - root supergroup          0 2017-09-02 02:11 /tmp

[root@master classes]#

[root@master classes]#

第二次没有删除package

package org.apache.hadoop.examples;

import java.io.IOException;

import java.util.StringTokenizer;

import org.apache.hadoop.conf.Configuration;

import org.apache.hadoop.fs.Path;

import org.apache.hadoop.io.IntWritable;

import org.apache.hadoop.io.Text;

import org.apache.hadoop.mapreduce.Job;

import org.apache.hadoop.mapreduce.Mapper;

import org.apache.hadoop.mapreduce.Reducer;

import org.apache.hadoop.mapreduce.lib.input.FileInputFormat;

import org.apache.hadoop.mapreduce.lib.output.FileOutputFormat;

import org.apache.hadoop.util.GenericOptionsParser;

public class WordCount {

  public static class TokenizerMapper

       extends Mapper<Object, Text, Text, IntWritable>{

    private final static IntWritable one = new IntWritable(1);

    private Text word = new Text();

    public void map(Object key, Text value, Context context

                    ) throws IOException, InterruptedException {

      StringTokenizer itr = new StringTokenizer(value.toString());

      while (itr.hasMoreTokens()) {

        word.set(itr.nextToken());

        context.write(word, one);

      }

    }

  }

  public static class IntSumReducer

       extends Reducer<Text,IntWritable,Text,IntWritable> {

    private IntWritable result = new IntWritable();

    public void reduce(Text key, Iterable<IntWritable> values,

                       Context context

                       ) throws IOException, InterruptedException {

      int sum = 0;

      for (IntWritable val : values) {

        sum += val.get();

      }

      result.set(sum);

      context.write(key, result);

    }

  }

  public static void main(String[] args) throws Exception {

    Configuration conf = new Configuration();

    String[] otherArgs = new GenericOptionsParser(conf, args).getRemainingArgs();

    if (otherArgs.length < 2) {

      System.err.println("Usage: wordcount <in> [<in>...] <out>");

      System.exit(2);

    }

    Job job = Job.getInstance(conf, "word count");

    job.setJarByClass(WordCount.class);

    job.setMapperClass(TokenizerMapper.class);

    job.setCombinerClass(IntSumReducer.class);

    job.setReducerClass(IntSumReducer.class);

    job.setOutputKeyClass(Text.class);

    job.setOutputValueClass(IntWritable.class);

    for (int i = 0; i < otherArgs.length - 1; ++i) {

      FileInputFormat.addInputPath(job, new Path(otherArgs[i]));

    }

    FileOutputFormat.setOutputPath(job,

      new Path(otherArgs[otherArgs.length - 1]));

    System.exit(job.waitForCompletion(true) ? 0 : 1);

  }

}

[root@master classes]#

[root@master classes]# tree

.

0 directories, 0 files

[root@master classes]# javac -classpath .:/home/jars/* -d /home/classes/ /home/javaFile/WordCount.java

[root@master classes]# tree

.

└── org

    └── apache

        └── hadoop

            └── examples

                ├── WordCount.class

                ├── WordCount$IntSumReducer.class

                └── WordCount$TokenizerMapper.class

4 directories, 3 files

[root@master classes]# jar -cvf wordcount.jar ./*

added manifest

adding: org/(in = 0) (out= 0)(stored 0%)

adding: org/apache/(in = 0) (out= 0)(stored 0%)

adding: org/apache/hadoop/(in = 0) (out= 0)(stored 0%)

adding: org/apache/hadoop/examples/(in = 0) (out= 0)(stored 0%)

adding: org/apache/hadoop/examples/WordCount$TokenizerMapper.class(in = 1790) (out= 764)(deflated 57%)

adding: org/apache/hadoop/examples/WordCount$IntSumReducer.class(in = 1793) (out= 749)(deflated 58%)

adding: org/apache/hadoop/examples/WordCount.class(in = 1988) (out= 1050)(deflated 47%)

[root@master classes]# /home/hadoop-2.7.4/bin/hadoop jar /home/classes/wordcount.jar org.apache.hadoop.examples.WordCount /hdfs-input.txt /result-package

17/09/02 02:20:41 INFO client.RMProxy: Connecting to ResourceManager at master/192.168.0.80:8032

17/09/02 02:20:43 INFO input.FileInputFormat: Total input paths to process : 1

17/09/02 02:20:43 INFO mapreduce.JobSubmitter: number of splits:1

17/09/02 02:20:43 INFO mapreduce.JobSubmitter: Submitting tokens for job: job_1504320356950_0011

17/09/02 02:20:43 INFO impl.YarnClientImpl: Submitted application application_1504320356950_0011

17/09/02 02:20:43 INFO mapreduce.Job: The url to track the job: http://master:8088/proxy/application_1504320356950_0011/

17/09/02 02:20:43 INFO mapreduce.Job: Running job: job_1504320356950_0011

17/09/02 02:20:51 INFO mapreduce.Job: Job job_1504320356950_0011 running in uber mode : false

17/09/02 02:20:51 INFO mapreduce.Job:  map 0% reduce 0%

17/09/02 02:20:58 INFO mapreduce.Job:  map 100% reduce 0%

17/09/02 02:21:05 INFO mapreduce.Job:  map 100% reduce 100%

17/09/02 02:21:06 INFO mapreduce.Job: Job job_1504320356950_0011 completed successfully

17/09/02 02:21:06 INFO mapreduce.Job: Counters: 49

    File System Counters

        FILE: Number of bytes read=118

        FILE: Number of bytes written=241857

        FILE: Number of read operations=0

        FILE: Number of large read operations=0

        FILE: Number of write operations=0

        HDFS: Number of bytes read=174

        HDFS: Number of bytes written=76

        HDFS: Number of read operations=6

        HDFS: Number of large read operations=0

        HDFS: Number of write operations=2

    Job Counters

        Launched map tasks=1

        Launched reduce tasks=1

        Data-local map tasks=1

        Total time spent by all maps in occupied slots (ms)=3828

        Total time spent by all reduces in occupied slots (ms)=4312

        Total time spent by all map tasks (ms)=3828

        Total time spent by all reduce tasks (ms)=4312

        Total vcore-milliseconds taken by all map tasks=3828

        Total vcore-milliseconds taken by all reduce tasks=4312

        Total megabyte-milliseconds taken by all map tasks=3919872

        Total megabyte-milliseconds taken by all reduce tasks=4415488

    Map-Reduce Framework

        Map input records=6

        Map output records=12

        Map output bytes=118

        Map output materialized bytes=118

        Input split bytes=98

        Combine input records=12

        Combine output records=9

        Reduce input groups=9

        Reduce shuffle bytes=118

        Reduce input records=9

        Reduce output records=9

        Spilled Records=18

        Shuffled Maps =1

        Failed Shuffles=0

        Merged Map outputs=1

        GC time elapsed (ms)=186

        CPU time spent (ms)=1200

        Physical memory (bytes) snapshot=297316352

        Virtual memory (bytes) snapshot=4159815680

        Total committed heap usage (bytes)=139595776

    Shuffle Errors

        BAD_ID=0

        CONNECTION=0

        IO_ERROR=0

        WRONG_LENGTH=0

        WRONG_MAP=0

        WRONG_REDUCE=0

    File Input Format Counters

        Bytes Read=76

    File Output Format Counters

        Bytes Written=76

[root@master classes]# /home/hadoop-2.7.4/bin/hadoop fs -ls /

Found 4 items

-rw-r--r--   2 root supergroup         76 2017-09-02 00:57 /hdfs-input.txt

drwxr-xr-x   - root supergroup          0 2017-09-02 02:21 /result-package

drwxr-xr-x   - root supergroup          0 2017-09-02 02:12 /result-self-compile

drwx------   - root supergroup          0 2017-09-02 02:11 /tmp

[root@master classes]#

[root@master classes]#

为啥要删除package，就是因为有包路径的时候调用方式就要 xxx.xxxxx.xxx来执行，而且打包的时候就不能只打class了，目录结构也要一并打进去

同理，自己写的代码也可按照这个方式执行

顺便提一点，如果只是打jar包用

jar -cvf test.jar XXX.class

但是如果要修改MANIFEST.MF，在里面指定mainClass，按照如下方式

#解压文件

jar -xf test.jar 

#在MANIFEST.MF 增加mainclass

Manifest-Version: 1.0

Created-By: 1.6.0_20 (Sun Microsystems Inc.)

Main-class: WordCount

#再打包

jar -cvfm test.jar MANIFEST.MF XXXX.class

这样就可以直接用 java -jar test.jar 运行了，后面不用跟具体的类

Linux巩固记录（5） hadoop 2.7.4下自己编译代码并运行MapReduce程序的更多相关文章

高可用Hadoop平台－运行MapReduce程序
1.概述最近有同学反应,如何在配置了HA的Hadoop平台运行MapReduce程序呢?对于刚步入Hadoop行业的同学,这个疑问却是会存在,其实仔细想想,如果你之前的语言功底不错的,应该会想到自动 ...
hive--构建于hadoop之上、让你像写SQL一样编写MapReduce程序
hive介绍什么是hive? hive:由Facebook开源用于解决海量结构化日志的数据统计 hive是基于hadoop的一个数据仓库工具,可以将结构化的数据映射为数据库的一张表,并提供类SQL查 ...
攻城狮在路上（陆）-- 配置hadoop本地windows运行MapReduce程序环境
本文的目的是实现在windows环境下实现模拟运行Map/Reduce程序.最终实现效果:MapReduce程序不会被提交到实际集群,但是运算结果会写入到集群的HDFS系统中. 一.环境说明: ...
hadoop——在命令行下编译并运行map-reduce程序 2
hadoop map-reduce程序的编译需要依赖hadoop的jar包,我尝试javac编译map-reduce时指定-classpath的包路径,但无奈hadoop的jar分布太散乱,根据自己 ...
Hadoop 系列文章(三) 配置部署启动YARN及在YARN上运行MapReduce程序
这篇文章里我们将用配置 YARN,在 YARN 上运行 MapReduce. 1.修改 yarn-env.sh 环境变量里的 JAVA_HOME 路径 [bamboo@hadoop-senior ha ...
Hadoop YARN上运行MapReduce程序
(1)配置集群 (a)配置hadoop-2.7.2/etc/hadoop/yarn-env.sh 配置一下JAVA_HOME export JAVA_HOME=/home/hadoop/bigdata ...
hadoop 编译代码及运行
搞定了hadoop配置之后,可以写代码运行了,首先要配一下CLASS_PATH,修改/etc/profile export JAVA_HOME=/usr/lib/jvm/java--openjdk-i ...
Hadoop日记Day16---命令行运行MapReduce程序
一.代码编写 1.1 单词统计回顾我们以前单词统计的例子,如代码1.1所示. package counter; import java.net.URI; import org.apache.hado ...
在window下远程虚拟机（centos）hadoop运行mapreduce程序
(注:虽然连接成功但是还是执行不了.以后有时间再解决吧看到的人别参考仅作个人笔记)先mark下 1.首先在window下载好一个eclipse.和拷贝好linux里面hadoop版本对应的插件(我是 ...

随机推荐

初学者问题一oracle
问:(待解决)如何将纵向表改成横向表? (待解决)如何实现对大型数据范围差距不大的索引?(建什么索引树)
POJ-3252 Avenger
题意:在区间中,他们化成2进制的数的0的个数大于等于1的数有多少个. 思路:我们需要记录上一次0和1的个数,此外我们还要特别注意一下前导0. 如果前面全是0的时候我们就要注意下一位是不是还是0,如果一 ...
python之初级篇2
一.数字类型 1)整数 int 类型 - bit_length() # 查询以二进制表示一个数字的值所需的位数 - int.from_bytes(bytes,byteorder) # 返回给定字节数组 ...
再读c++primer plus 002
1.读取char值时,与读取其它基本类型一样,cin将忽略空格和换行符,函数cin.get(ch)读取输入的下一个字符(即使是空格),并将其赋给变量ch. 2.指针和const:(1)让指针指向一个常 ...
java测试ATM自助操作系统
开学第一周系主任安排了一项测试,测试要求:模拟ATM自助取款机用文件进行存储账户信息,密码等,并进行存款取款,转账,查询记录等操作,而且要进行文件的读取与录入. 这是一个ATM自助取款的操作系统,进行 ...
2018.10.31 bzoj3339&&3585mex（主席树）
传送门双倍经验直接上主席树,每个叶节点维护这个值出现的最右区间,非叶子节点维护当前值域内所有最右区间的最小值. 查询的时候只用在以root[qr]root[qr]root[qr]为根的树上面二分. ...
android 应用商店
下面更多 http://wiki.youmi.net/Wiki/PromotionChannelIDs 小米 http://market.xiaomi.com/dev安智市场 http://dev.a ...
vue 开发系列（七）路由配置
概要用 Vue.js + vue-router 创建单页应用,是非常简单的.使用 Vue.js ,我们已经可以通过组合组件来组成应用程序,当你要把 vue-router 添加进来,我们需要做的是,将 ...
mysql知识积累
验证mysql工作状态 systemctl status mysql.service 启动 sudo systemctl start mysql 停止 service mysql stop 重启mys ...
idea关于tab的设置
新手使用,一不小心tab显示在右面了,这不学习下给搞正常点. settings===>Editor=====>Editor Tabs; Palacement设置的是tab显示的部位: Ta ...

Linux巩固记录（5） hadoop 2.7.4下自己编译代码并运行MapReduce程序

Linux巩固记录（5） hadoop 2.7.4下自己编译代码并运行MapReduce程序的更多相关文章

随机推荐

热门专题