黑猴子的家:HBase 自定義HBase-MapReduce案列一

weixin_34234823發表於2018-10-05

將fruit表中的一部分資料,通過MR遷入到fruit_mr表中

1、Code -> GitHub

https://github.com/liufengji/hbase_mapredece_one.git

2、構建ReadFruitMapper類,用於讀取fruit表中的資料

import java.io.IOException;
import org.apache.hadoop.hbase.Cell;
import org.apache.hadoop.hbase.CellUtil;
import org.apache.hadoop.hbase.client.Put;
import org.apache.hadoop.hbase.client.Result;
import org.apache.hadoop.hbase.io.ImmutableBytesWritable;
import org.apache.hadoop.hbase.mapreduce.TableMapper;
import org.apache.hadoop.hbase.util.Bytes;

public class ReadFruitMapper extends TableMapper<ImmutableBytesWritable, Put> {

    @Override
    protected void map(ImmutableBytesWritable key, Result value, Context context) 
    throws IOException, InterruptedException {
    //將fruit的name和color提取出來,相當於將每一行資料讀取出來放入到Put物件中。
        Put put = new Put(key.get());
        //遍歷新增column行
        for(Cell cell: value.rawCells()){
            //新增/克隆列族:info
            if("info".equals(Bytes.toString(CellUtil.cloneFamily(cell)))){
                //新增/克隆列:name
                if("name".equals(Bytes.toString(CellUtil.cloneQualifier(cell)))){
                    //將該列cell加入到put物件中
                    put.add(cell);
                    //新增/克隆列:color
                }else if("color".equals(
                                 Bytes.toString(CellUtil.cloneQualifier(cell)))){
                    //向該列cell加入到put物件中
                    put.add(cell);
                }
            }
        }
        //將從fruit讀取到的每行資料寫入到context中作為map的輸出
        context.write(key, put);
    }
}

3、構建WriteFruitMRReducer類,用於將讀取到的fruit表中的資料寫入到fruit_mr表中

import java.io.IOException;
import org.apache.hadoop.hbase.client.Put;
import org.apache.hadoop.hbase.io.ImmutableBytesWritable;
import org.apache.hadoop.hbase.mapreduce.TableReducer;
import org.apache.hadoop.io.NullWritable;


public class WriteFruitMRReducer extends TableReducer<ImmutableBytesWritable,
                                                             Put, NullWritable> {
    @Override
    protected void reduce(ImmutableBytesWritable key, Iterable<Put> values,
                                                    Context context) 
    throws IOException, InterruptedException {
        //讀出來的每一行資料寫入到fruit_mr表中
        for(Put put: values){
            context.write(NullWritable.get(), put);
        }
    }
}

4、構建Fruit2FruitMRRunner用於組裝執行Job任務


import java.io.IOException;

import org.apache.hadoop.conf.Configuration;
import org.apache.hadoop.conf.Configured;
import org.apache.hadoop.hbase.HBaseConfiguration;
import org.apache.hadoop.hbase.client.Put;
import org.apache.hadoop.hbase.client.Scan;
import org.apache.hadoop.hbase.io.ImmutableBytesWritable;
import org.apache.hadoop.hbase.mapreduce.TableMapReduceUtil;
import org.apache.hadoop.mapreduce.Job;
import org.apache.hadoop.util.Tool;
import org.apache.hadoop.util.ToolRunner;

public class Fruit2FruitMRRunner extends Configured implements Tool {

    @Override
    public int run(String[] arg0) throws Exception {

        // 得到Configuration
        Configuration conf = this.getConf();

        // 建立Job任務
        Job job = Job.getInstance(conf, this.getClass().getSimpleName());
        job.setJarByClass(Fruit2FruitMRRunner.class);

        // 配置Job
        Scan scan = new Scan();
        scan.setCacheBlocks(false);
        scan.setCaching(500);

        // 設定Mapper,注意匯入的是mapreduce包下的,不是mapred包下的,後者是老版本
        TableMapReduceUtil.initTableMapperJob("fruit", // 資料來源的表名
                scan, // scan掃描控制器
                ReadFruitMapper.class, // 設定Mapper類
                ImmutableBytesWritable.class, // 設定Mapper輸出key型別
                Put.class, // 設定Mapper輸出value值型別
                job// 設定給哪個JOB
        );

        // 設定Reducer
        TableMapReduceUtil.initTableReducerJob("fruit_mr",
                                             WriteFruitMRReducer.class, job);

        // 設定Reduce數量,最少1個
        job.setNumReduceTasks(1);

        boolean isSuccess = job.waitForCompletion(true);
        if (!isSuccess) {
            throw new IOException("Job running with error");
        }
        return isSuccess ? 0 : 1;
    }

}

5、主函式中呼叫執行該Job任務

    public static void main(String[] args) throws Exception {
        Configuration conf = HBaseConfiguration.create();
        int status = ToolRunner.run(conf, new Fruit2FruitMRRunner(), args);
        System.exit(status);
    }

6、打包執行任務

[victor@node1 hbase-1.3.1]$ /opt/module/hadoop-2.7.2/bin/yarn jar \
hbase-0.0.1-SNAPSHOT.jar com.victor.hbase.mr1.Fruit2FruitMRRunner

尖叫提示:執行任務前,如果待資料匯入的表不存在,則需要提前建立之。
尖叫提示:maven打包命令:-P local clean package或-P dev clean package install(將第三方jar包一同打包,需要外掛:maven-shade-plugin)

相關文章