8
I have a simple hadoop application, which get one CSV file, then split the entry by ",", then count the first items.
The following is my code.
package com.bluedolphin;
import java.io.IOException;
import java.util.Iterator;
import org.apache.hadoop.conf.Configuration;
import org.apache.hadoop.conf.Configured;
import org.apache.hadoop.fs.Path;
import org.apache.hadoop.io.IntWritable;
import org.apache.hadoop.io.LongWritable;
import org.apache.hadoop.io.Text;
import org.apache.hadoop.mapred.OutputCollector;
import org.apache.hadoop.mapred.Reporter;
import org.apache.hadoop.mapreduce.Job;
import org.apache.hadoop.mapreduce.Mapper;
import org.apache.hadoop.mapreduce.Reducer;
import org.apache.hadoop.mapreduce.lib.input.FileInputFormat;
import org.apache.hadoop.mapreduce.lib.input.TextInputFormat;
import org.apache.hadoop.mapreduce.lib.output.FileOutputFormat;
import org.apache.hadoop.util.Tool;
import org.apache.hadoop.util.ToolRunner;
public class MyJob extends Configured implements Tool {
private final static LongWritable one = new LongWritable(1);
public static class MapClass extends Mapper<Object, Text, Text, LongWritable> {
private Text word = new Text();
public void map(Object key,
Text value,
OutputCollector<Text, LongWritable> output,
Reporter reporter) throws IOException, InterruptedException {
String[] citation = value.toString().split(",");
word.set(citation[0]);
output.collect(word, one);
}
}
public static class Reduce extends Reducer<Text, LongWritable, Text, LongWritable> {
public void reduce(
Text key,
Iterator<LongWritable> values,
OutputCollector<Text, LongWritable> output,
Reporter reporter) throws IOException, InterruptedException {
int sum = 0;
while (values.has