7.1 MapReduce(上) · Hadoop

參考文獻：https://blog.csdn.net/markcheney/article/details/53998796 前言： MapReduce是一個高性能的批處理分布式計算框架，用于對海量數據進行并行分析和處理。與傳統方法相比較，MapReduce更傾向于蠻力去解決問題，通過簡單、粗暴、有效的方式去處理海量的數據。通過對數據的輸入、拆分與組合（核心），將任務分配到多個節點服務器上，進行分布式計算，這樣可以有效地提高數據管理的安全性，同時也能夠很好地范圍被管理的數據。 mapreduce概念+實例 ![](https://box.kancloud.cn/9f108208b93a5e84cc9b47c5cf5e1abb_735x301.jpg) mapreduce核心就是map+shuffle+reducer，首先通過讀取文件，進行分片，通過map獲取文件的key-value映射關系，用作reducer的輸入，在作為reducer輸入之前，要先對map的key進行一個shuffle，也就是排個序，然后將排完序的key-value作為reducer的輸入進行reduce操作，當然一個mapreduce任務可以不要有reduce，只用一個map，接下來就來講解一個mapreduce界的“hello world”。 ``` import java.io.IOException; import java.util.StringTokenizer; import org.apache.hadoop.conf.Configuration; import org.apache.hadoop.fs.Path; import org.apache.hadoop.io.IntWritable; import org.apache.hadoop.io.Text; import org.apache.hadoop.mapreduce.Job; import org.apache.hadoop.mapreduce.Mapper; import org.apache.hadoop.mapreduce.Reducer; import org.apache.hadoop.mapreduce.lib.input.FileInputFormat; import org.apache.hadoop.mapreduce.lib.output.FileOutputFormat; public class WordCount { public static class TokenizerMapper extends Mapper<Object, Text, Text, IntWritable>{ private final static IntWritable one = new IntWritable(1); private Text word = new Text(); public void map(Object key, Text value, Context context ) throws IOException, InterruptedException { System.out.println("key=:"+key.toString()); System.out.println("value=:"+value.toString()); StringTokenizer itr = new StringTokenizer(value.toString()); while (itr.hasMoreTokens()) { word.set(itr.nextToken()); context.write(word, one); } System.out.println("context=:"+context.toString()); } } public static class IntSumReducer extends Reducer<Text,IntWritable,Text,IntWritable> { private IntWritable result = new IntWritable(); public void reduce(Text key, Iterable<IntWritable> values, Context context) throws IOException, InterruptedException { int sum = 0; for (IntWritable val : values) { sum += val.get(); } result.set(sum); context.write(key, result); } } public static void main(String[] args) throws Exception { System.setProperty("hadoop.home.dir", "D:\\hadoop2.7.6"); Configuration conf = new Configuration(); Job job = new Job(conf); job.setJarByClass(WordCount.class); job.setMapperClass(TokenizerMapper.class); job.setReducerClass(IntSumReducer.class); job.setMapOutputKeyClass(Text.class); job.setMapOutputValueClass(IntWritable.class); job.setOutputKeyClass(Text.class); job.setOutputValueClass(IntWritable.class); FileInputFormat.addInputPath(job, new Path("d:/input4/WordCount.java")); FileOutputFormat.setOutputPath(job, new Path("d:/output4")); job.waitForCompletion(true); } } ```