mutilple output reduce cannot write

package org.lukey.hadoop.classifyBayes;

import java.io.BufferedReader;

import java.io.IOException;

import java.io.InputStreamReader;

import java.net.URI;

import org.apache.hadoop.conf.Configuration;

import org.apache.hadoop.fs.FSDataInputStream;

import org.apache.hadoop.fs.FileSystem;

import org.apache.hadoop.fs.Path;

import org.apache.hadoop.io.DoubleWritable;

import org.apache.hadoop.io.IntWritable;

import org.apache.hadoop.io.LongWritable;

import org.apache.hadoop.io.Text;

import org.apache.hadoop.mapreduce.Job;

import org.apache.hadoop.mapreduce.Mapper;

import org.apache.hadoop.mapreduce.Reducer;

import org.apache.hadoop.mapreduce.lib.input.FileInputFormat;

import org.apache.hadoop.mapreduce.lib.output.FileOutputFormat;

import org.apache.hadoop.mapreduce.lib.output.MultipleOutputs;

public class Probability {

    // Client

    public static void main(String[] args) throws Exception {

        Configuration conf = new Configuration();

        //读取单词总数，设置到congfiguration中

        String totalWordsPath = "/user/hadoop/output/totalwords.txt";

        FileSystem fs = FileSystem.get(URI.create(totalWordsPath), conf);

        FSDataInputStream inputStream = fs.open(new Path(totalWordsPath));

        BufferedReader buffer = new BufferedReader(new InputStreamReader(inputStream));

        String strLine = buffer.readLine();

        String[] temp = strLine.split(":");

        if(temp.length == 2){

            //temp[0] = TOTALWORDS

            conf.setInt(temp[0], Integer.parseInt(temp[1]));

        }

        /*

        String[] otherArgs = new GenericOptionsParser(conf, args).getRemainingArgs();

        if (otherArgs.length != 2) {

            System.out.println("Usage <in> <out>");

            System.exit(-1);

        }

*/

        Job job = new Job(conf, "file count");

        job.setJarByClass(Probability.class);

        job.setMapperClass(WordsOfClassCountMapper.class);

        job.setReducerClass(WordsOfClassCountReducer.class);

        String input = "/user/hadoop/mid/wordsFrequence";

        String output = "/user/hadoop/output/probability/";

        FileInputFormat.addInputPath(job, new Path(input));

        FileOutputFormat.setOutputPath(job, new Path(output));

        job.setOutputKeyClass(Text.class);

        job.setOutputValueClass(IntWritable.class);

        System.exit(job.waitForCompletion(true) ? 0 : 1);

    }

    private static MultipleOutputs<Text, IntWritable> mos;

    // Mapper

    static class WordsOfClassCountMapper extends Mapper<LongWritable, Text, Text, IntWritable> {

        private  static IntWritable number = new IntWritable();

        @Override

        protected void map(LongWritable key, Text value, Mapper<LongWritable, Text, Text, IntWritable>.Context context)

                throws IOException, InterruptedException {

            String[] temp = value.toString().split("\t");

            if(temp.length == 3){

                // 文件夹名类别名

                String dirName = temp[0];

                value.set(temp[1]);

                number.set(Integer.parseInt(temp[2]));

                mos.write(value, number, dirName);

            }

        }

        @Override

        protected void cleanup(Mapper<LongWritable, Text, Text, IntWritable>.Context context)

                throws IOException, InterruptedException {

            // TODO Auto-generated method stub

            mos.close();

        }

        @Override

        protected void setup(Mapper<LongWritable, Text, Text, IntWritable>.Context context)

                throws IOException, InterruptedException {

            // TODO Auto-generated method stub

            mos = new MultipleOutputs<Text, IntWritable>(context);

        }

    }

    // Reducer

    static class WordsOfClassCountReducer extends Reducer<Text, IntWritable, Text, DoubleWritable> {

        // result 表示每个文件里面单词个数

        DoubleWritable result = new DoubleWritable(3);

        Configuration conf = new Configuration();

        int total = conf.getInt("TOTALWORDS", 1);

        @Override

        protected void reduce(Text key, Iterable<IntWritable> values,

                Reducer<Text, IntWritable, Text, DoubleWritable>.Context context)

                        throws IOException, InterruptedException {

            // TODO Auto-generated method stub

//            double sum = 0;

//            for (IntWritable value : values) {

//                sum += value.get();

//            }

//            result.set(sum);

            context.write(key, result);

        }

    }

}

mutilple output reduce cannot write的更多相关文章

2019.12.05【ABAP随笔】分组循环(LOOP AT Group) / REDUCE
ABAP 7.40新语法 LOOP AT Group 和 REDUCE *LOOP AT itab result [cond] GROUP BY key ( key1 = dobj1 key2 = d ...
Hadoop基础概念介绍
基于YARN的配置信息, 参见: http://www.ibm.com/developerworks/cn/opensource/os-cn-hadoop-yarn/ hadoop入门 - 基础概念 ...
MapReduce执行流程及程序编写
MapReduce 一种分布式计算模型,解决海量数据的计算问题,MapReduce将计算过程抽象成两个函数 Map(映射):对一些独立元素(拆分后的小块)组成的列表的每一个元素进行指定的操作,可以高度 ...
（3）Deep Learning之神经网络和反向传播算法
往期回顾在上一篇文章中,我们已经掌握了机器学习的基本套路,对模型.目标函数.优化算法这些概念有了一定程度的理解,而且已经会训练单个的感知器或者线性单元了.在这篇文章中,我们将把这些单独的单元按照一定 ...
javaScript系列 [09]-javaScript和JSON (拓展)
本文输出JSON搜索和JSON转换相关的内容,是对前两篇文章的补充. JSON搜索在特定的开发场景中,如果服务器端返回的JSON数据异常复杂(可能超过上万行),那么必然就有对JSON文档进行搜索的需 ...
Hadoop源码分析（mapreduce.lib.partition/reduce/output）
Map的结果,会通过partition分发到Reducer上.Reducer做完Reduce操作后,通过OutputFormat,进行输出.以下我们就来分析參与这个过程的类. Mapper的结果, ...
MapReduce剖析笔记之七：Child子进程处理Map和Reduce任务的主要流程
在上一节我们分析了TaskTracker如何对JobTracker分配过来的任务进行初始化,并创建各类JVM启动所需的信息,最终创建JVM的整个过程,本节我们继续来看,JVM启动后,执行的是Child ...
MapReduce剖析笔记之三：Job的Map/Reduce Task初始化
上一节分析了Job由JobClient提交到JobTracker的流程,利用RPC机制,JobTracker接收到Job ID和Job所在HDFS的目录,够早了JobInProgress对象,丢入队列 ...
【hadoop】如何向map和reduce脚本传递参数,加载文件和目录
本文主要讲解三个问题: 1 使用Java编写MapReduce程序时,如何向map.reduce函数传递参数. 2 使用Streaming编写MapReduce程序(C/C++ ...

随机推荐

《Windows驱动开发技术详解》之读写操作
缓冲区方式读写操作设置缓冲区读写方式:
Chapter 2 Open Book——18
"Wow," Mike said. "It's snowing."I looked at the little cotton fluffs that were ...
leetcode387
Given a string, find the first non-repeating character in it and return it's index. If it doesn't ex ...
qsdk编译
QSDK是一种在openwrt的基础上,加入了高通atheros芯片相关资料的一种环境. QSDK与openwrt的区别主要在如下几个方面: arch/mips/ath79/* – updated Q ...
Beanstalkd
摘要by ck:beanstalkd 和 kafka的本质区别是什么? Beanstalkd,一个高性能.轻量级的分布式内存队列系统,最初设计的目的是想通过后台异步执行耗时的任务来降低高容量Web ...
A*搜寻算法（A星算法）
A*搜寻算法[编辑] 维基百科,自由的百科全书本条目需要补充更多来源.(2015年6月30日) 请协助添加多方面可靠来源以改善这篇条目,无法查证的内容可能会被提出异议而移除. A*搜索算法,俗称A星 ...
php根据时间显示刚刚,几分钟前,今天,昨天的实现代码
如果大家有更好的方案欢迎交流 function diffBetweenTwoDay($pastDay){ $timeC = time() - strtotime($pastDay); $dateC = ...
asp.net导出excel科学计数问题
方法一: 在asp.net 中我一般都是将要导出的数据放到gridview网格里,首先对网格邦定数据时字符串形式处理,然后再用普通的形式导出excel就把问题解决了. 我的代码非常简单:在邦定gr ...
js操作select和option
1.动态创建select function createSelect(){ var mySelect = document.createElement_x("select"); m ...
CodeForces 719B Anatoly and Cockroaches 思维锻炼题
题目大意:有一排蟑螂,只有r和b两种颜色,你可以交换任意两只蟑螂的位置,或涂改一个蟑螂的颜色,使其变成r和b交互排列的形式.问做少的操作次数. 题目思路:更改后的队列只有两种形式:长度为n以r开头:长 ...

mutilple output reduce cannot write

mutilple output reduce cannot write的更多相关文章

随机推荐

热门专题