前言
在使用MapReduce对数据进行处理的过程中,难免会遇到在一个文件夹中避开某类文件的问题,在本篇博客中我们使用PathFilter路径过滤器过滤掉*.txt文件。这里使用词频统计来做一个简单的小例子.
一、定义Mapper类
import org.apache.hadoop.io.IntWritable;
import org.apache.hadoop.io.LongWritable;
import org.apache.hadoop.io.Text;
import org.apache.hadoop.mapreduce.Mapper;
import java.io.IOException;
public class WcMapper extends Mapper<LongWritable, Text,Text, IntWritable> {
Text k = new Text();
IntWritable v = new IntWritable(1);
@Override
protected void map(LongWritable key, Text value, Context context) throws IOException, InterruptedException {
String line = value.toString();
String[] words = line.split(" ");
for (String word : words) {
k