hadoop-数据去重

import java.io.IOException;
import java.util.StringTokenizer;

import org.apache.hadoop.conf.Configuration;
import org.apache.hadoop.fs.Path;
import org.apache.hadoop.io.IntWritable;
import org.apache.hadoop.io.Text;
import org.apache.hadoop.mapreduce.Job;
import org.apache.hadoop.mapreduce.Mapper;
import org.apache.hadoop.mapreduce.Reducer;
import org.apache.hadoop.mapreduce.lib.input.FileInputFormat;
import org.apache.hadoop.mapreduce.lib.output.FileOutputFormat;
import org.apache.hadoop.util.GenericOptionsParser;

public class Dedup {
	//map将输入中的value复制到输出数据的key上,并直接输出
	public static class Map extends Mapper{
		private static Text line = new Text();
		public void map(Object key,Text value,Context context) throws IOException,InterruptedException{
			line = value;
			context.write(line, new Text(""));
		}
	}
	//reduce将输入中的key复制到输出数据的key上,并直接输出
	public static class Reduce extends Reducer{
		public void reduce(Text key,Iterable values,Context context) throws IOException,InterruptedException{
			context.write(key, new Text(""));
			
		}
	}
	/**
	 * @param args
	 */
	public static void main(String[] args) throws Exception{
		// TODO Auto-generated method stub
		Configuration conf = new Configuration();
		String[] otherArgs = new GenericOptionsParser(conf,args).getRemainingArgs();
		if(otherArgs.length != 2){
			System.err.println("Usage WordCount  ");
			System.exit(2);
		}
		Job job = new Job(conf,"Dedup");
		job.setJarByClass(Dedup.class);
		job.setMapperClass(Map.class);
		job.setCombinerClass(Reduce.class);
		job.setReducerClass(Reduce.class);
		job.setOutputKeyClass(Text.class);
		job.setOutputValueClass(Text.class);
		FileInputFormat.addInputPath(job, new Path(otherArgs[0]));
		FileOutputFormat.setOutputPath(job, new Path(otherArgs[1]));
		System.exit(job.waitForCompletion(true) ? 0 : 1);
	}

}


在上传到linux下的话,假如编译不了的话,可把注释的中文去掉。

把该文件上传到我的hadoop安装目录的firstProject目录下

首先编译该文件

javac -classpath hadoop-core-1.1.2.jar:/opt/hadoop-1.1.2/lib/commons-cli-1.2.jar -d firstProject firstProject/Dedup.java


之后打包

 jar -cvf dedup.jar -C firstProject/ .


可在安装目录看到dedup.jar

新建两个文件

text01
2006-6-10 b
2006-6-11 c
2006-6-12 d
2006-6-13 a
2006-6-14 b
2006-6-15 c
2006-6-11 c
text02
2006-6-9 b
2006-6-10 b
2006-6-11 b
2006-6-12 d
2006-6-13 a
2006-6-14 c
2006-6-15 d
2006-6-11 c


上传

hadoop fs -put /opt/hadoop-1.1.2/text0* input  


运行

hadoop jar dedup.jar Dedup input output3


查看结果

hadoop fs -cat output3/part-r-00000


 

hadoop dfsadmin -safemode leave


关闭安全模式

你可能感兴趣的:(hadoop)