hadoop-数据去重

最新推荐文章于 2024-05-11 19:12:00 发布

a331251021

最新推荐文章于 2024-05-11 19:12:00 发布

阅读量1.6k

点赞数

分类专栏： hadoop

本文链接：https://blog.csdn.net/tianba8/article/details/9672891

版权

hadoop 专栏收录该内容

4 篇文章 0 订阅

订阅专栏

import java.io.IOException;
import java.util.StringTokenizer;

import org.apache.hadoop.conf.Configuration;
import org.apache.hadoop.fs.Path;
import org.apache.hadoop.io.IntWritable;
import org.apache.hadoop.io.Text;
import org.apache.hadoop.mapreduce.Job;
import org.apache.hadoop.mapreduce.Mapper;
import org.apache.hadoop.mapreduce.Reducer;
import org.apache.hadoop.mapreduce.lib.input.FileInputFormat;
import org.apache.hadoop.mapreduce.lib.output.FileOutputFormat;
import org.apache.hadoop.util.GenericOptionsParser;

public class Dedup {
	//map将输入中的value复制到输出数据的key上，并直接输出
	public static class Map extends Mapper<Object,Text,Text,Text>{
		private static Text line = new Text();
		public void map(Object key,Text value,Context context) throws IOException,InterruptedException{
			line = value;
			context.write(line, new Text(""));
		}
	}
	//reduce将输入中的key复制到输出数据的key上，并直接输出
	public static class Reduce extends Reducer<Text,Text,Text,Text>{
		public void reduce(Text key,Iterable<Text> values,Context context) throws IOException,InterruptedException{
			context.write(key, new Text(""));
			
		}
	}
	/**
	 * @param args
	 */
	public static void main(String[] args) throws Exception{
		// TODO Auto-generated method stub
		Configuration conf = new Configuration();
		String[] otherArgs = new GenericOptionsParser(conf,args).getRemainingArgs();
		if(otherArgs.length != 2){
			System.err.println("Usage WordCount <int> <out>");
			System.exit(2);
		}
		Job job = new Job(conf,"Dedup");
		job.setJarByClass(Dedup.class);
		job.setMapperClass(Map.class);
		job.setCombinerClass(Reduce.class);
		job.setReducerClass(Reduce.class);
		job.setOutputKeyClass(Text.class);
		job.setOutputValueClass(Text.class);
		FileInputFormat.addInputPath(job, new Path(otherArgs[0]));
		FileOutputFormat.setOutputPath(job, new Path(otherArgs[1]));
		System.exit(job.waitForCompletion(true) ? 0 : 1);
	}

}

在上传到linux下的话，假如编译不了的话，可把注释的中文去掉。

把该文件上传到我的hadoop安装目录的firstProject目录下

首先编译该文件

javac -classpath hadoop-core-1.1.2.jar:/opt/hadoop-1.1.2/lib/commons-cli-1.2.jar -d firstProject firstProject/Dedup.java

之后打包

 jar -cvf dedup.jar -C firstProject/ .

可在安装目录看到dedup.jar

新建两个文件

text01
2006-6-10 b
2006-6-11 c
2006-6-12 d
2006-6-13 a
2006-6-14 b
2006-6-15 c
2006-6-11 c
text02
2006-6-9 b
2006-6-10 b
2006-6-11 b
2006-6-12 d
2006-6-13 a
2006-6-14 c
2006-6-15 d
2006-6-11 c

上传

hadoop fs -put /opt/hadoop-1.1.2/text0* input

运行

hadoop jar dedup.jar Dedup input output3

查看结果

hadoop fs -cat output3/part-r-00000

hadoop dfsadmin -safemode leave

关闭安全模式

a331251021

关注

0
点赞
踩
2

收藏

觉得还不错? 一键收藏
0
评论
hadoop-数据去重

import java.io.IOException;import java.util.StringTokenizer;import org.apache.hadoop.conf.Configuration;import org.apache.hadoop.fs.Path;import org.apache.hadoop.io.IntWritable;import org.
复制链接

扫一扫