用spark导入数据到hbase

集群环境：一主三从，Spark为Spark On YARN模式

Spark导入hbase数据方式有多种

1.少量数据：直接调用hbase API的单条或者批量方法就可以

2.导入的数据量比较大，那就需要先生成hfile文件，在把hfile文件加载到hbase里面

下面主要介绍第二种方法：

该方法主要使用spark Java API的两个方法：

1.textFile：将本地文件或者HDFS文件转换成RDD

2.flatMapToPair：将每行数据的所有key-value对象合并成Iterator对象返回（针对多family，多column）

代码如下：

package scala;

import java.util.ArrayList;

import java.util.Iterator;

import java.util.List;

import org.apache.hadoop.conf.Configuration;

import org.apache.hadoop.fs.FileSystem;

import org.apache.hadoop.fs.Path;

import org.apache.hadoop.hbase.HBaseConfiguration;

import org.apache.hadoop.hbase.KeyValue;

import org.apache.hadoop.hbase.TableName;

import org.apache.hadoop.hbase.client.Admin;

import org.apache.hadoop.hbase.client.Connection;

import org.apache.hadoop.hbase.client.ConnectionFactory;

import org.apache.hadoop.hbase.client.Table;

import org.apache.hadoop.hbase.io.ImmutableBytesWritable;

import org.apache.hadoop.hbase.mapreduce.HFileOutputFormat2;

import org.apache.hadoop.hbase.mapreduce.LoadIncrementalHFiles;

import org.apache.hadoop.hbase.util.Bytes;

import org.apache.hadoop.mapreduce.Job;

import org.apache.hadoop.mapreduce.lib.input.FileInputFormat;

import org.apache.hadoop.mapreduce.lib.input.TextInputFormat;

import org.apache.hadoop.mapreduce.lib.output.FileOutputFormat;

import org.apache.spark.SparkConf;

import org.apache.spark.api.java.JavaPairRDD;

import org.apache.spark.api.java.JavaRDD;

import org.apache.spark.api.java.JavaSparkContext;

import org.apache.spark.api.java.function.PairFlatMapFunction;

import org.apache.spark.storage.StorageLevel;

import util.HFileLoader;

public class HbaseBulkLoad {

    private static final String ZKconnect="slave1,slave2,slave3:2181";

    private static final String HDFS_ADDR="hdfs://master:8020";

    private static final String TABLE_NAME="DBSTK.STKFSTEST";//表名

    private static final String COLUMN_FAMILY="FS";//列族

    public static void run(String[] args) throws Exception {

        Configuration configuration = HBaseConfiguration.create();

        configuration.set("hbase.zookeeper.quorum", ZKconnect);

        configuration.set("fs.defaultFS", HDFS_ADDR);

        configuration.set("dfs.replication", "1");

        String inputPath = args[0];

        String outputPath = args[1];

        Job job = Job.getInstance(configuration, "Spark Bulk Loading HBase Table:" + TABLE_NAME);

        job.setInputFormatClass(TextInputFormat.class);

        job.setMapOutputKeyClass(ImmutableBytesWritable.class);//指定输出键类

        job.setMapOutputValueClass(KeyValue.class);//指定输出值类

        job.setOutputFormatClass(HFileOutputFormat2.class);

        FileInputFormat.addInputPaths(job, inputPath);//输入路径

        FileSystem fs = FileSystem.get(configuration);

        Path output = new Path(outputPath);

        if (fs.exists(output)) {

            fs.delete(output, true);//如果输出路径存在，就将其删除

        }

        fs.close();

        FileOutputFormat.setOutputPath(job, output);//hfile输出路径

        //初始化sparkContext

        SparkConf sparkConf = new SparkConf().setAppName("HbaseBulkLoad").setMaster("local[*]");

        JavaSparkContext jsc = new JavaSparkContext(sparkConf);

        //读取数据文件

        JavaRDD<String> lines = jsc.textFile(inputPath);

        lines.persist(StorageLevel.MEMORY_AND_DISK_SER());

        JavaPairRDD<ImmutableBytesWritable,KeyValue> hfileRdd =

                lines.flatMapToPair(new PairFlatMapFunction<String, ImmutableBytesWritable, KeyValue>() {

            private static final long serialVersionUID = 1L;

            @Override

            public Iterator<Tuple2<ImmutableBytesWritable, KeyValue>> call(String text) throws Exception {

                List<Tuple2<ImmutableBytesWritable, KeyValue>> tps = new ArrayList<Tuple2<ImmutableBytesWritable, KeyValue>>();

                if(null == text || text.length()<1){

                    return tps.iterator();//不能返回null

                }

                String[] resArr = text.split(",");

                if(resArr != null && resArr.length == 14){

                    byte[] rowkeyByte = Bytes.toBytes(resArr[0]+resArr[3]+resArr[4]+resArr[5])

                    byte[] columnFamily = Bytes.toBytes(COLUMN_FAMILY);

                    ImmutableBytesWritable ibw = new ImmutableBytesWritable(rowkeyByte);

                    //EP,HP,LP,MK,MT,SC,SN,SP,ST,SY,TD,TM,TQ,UX（字典顺序排序）

                    //注意，这地方rowkey、列族和列都要按照字典排序，如果有多个列族，也要按照字典排序，rowkey排序我们交给spark的sortByKey去管理

                    tps.add(new Tuple2<>(ibw,new KeyValue(rowkeyByte, columnFamily, Bytes.toBytes("EP"),Bytes.toBytes(resArr[9]))));

                    tps.add(new Tuple2<>(ibw,new KeyValue(rowkeyByte, columnFamily, Bytes.toBytes("HP"),Bytes.toBytes(resArr[7]))));

                    tps.add(new Tuple2<>(ibw,new KeyValue(rowkeyByte, columnFamily, Bytes.toBytes("LP"),Bytes.toBytes(resArr[8]))));

                    tps.add(new Tuple2<>(ibw,new KeyValue(rowkeyByte, columnFamily, Bytes.toBytes("MK"),Bytes.toBytes(resArr[13]))));

                    tps.add(new Tuple2<>(ibw,new KeyValue(rowkeyByte, columnFamily, Bytes.toBytes("MT"),Bytes.toBytes(resArr[4]))));

                    tps.add(new Tuple2<>(ibw,new KeyValue(rowkeyByte, columnFamily, Bytes.toBytes("SC"),Bytes.toBytes(resArr[0]))));

                    tps.add(new Tuple2<>(ibw,new KeyValue(rowkeyByte, columnFamily, Bytes.toBytes("SN"),Bytes.toBytes(resArr[1]))));

                    tps.add(new Tuple2<>(ibw,new KeyValue(rowkeyByte, columnFamily, Bytes.toBytes("SP"),Bytes.toBytes(resArr[6]))));

                    tps.add(new Tuple2<>(ibw,new KeyValue(rowkeyByte, columnFamily, Bytes.toBytes("ST"),Bytes.toBytes(resArr[5]))));

                    tps.add(new Tuple2<>(ibw,new KeyValue(rowkeyByte, columnFamily, Bytes.toBytes("SY"),Bytes.toBytes(resArr[2]))));

                    tps.add(new Tuple2<>(ibw,new KeyValue(rowkeyByte, columnFamily, Bytes.toBytes("TD"),Bytes.toBytes(resArr[3]))));

                    tps.add(new Tuple2<>(ibw,new KeyValue(rowkeyByte, columnFamily, Bytes.toBytes("TM"),Bytes.toBytes(resArr[11]))));

                    tps.add(new Tuple2<>(ibw,new KeyValue(rowkeyByte, columnFamily, Bytes.toBytes("TQ"),Bytes.toBytes(resArr[10]))));

                    tps.add(new Tuple2<>(ibw,new KeyValue(rowkeyByte, columnFamily, Bytes.toBytes("UX"),Bytes.toBytes(resArr[12]))));

                }

                return tps.iterator();

            }

        }).sortByKey();

        Connection connection = ConnectionFactory.createConnection(configuration);

        TableName tableName = TableName.valueOf(TABLE_NAME);

        HFileOutputFormat2.configureIncrementalLoad(job, connection.getTable(tableName), connection.getRegionLocator(tableName));

        //生成hfile文件

        hfileRdd.saveAsNewAPIHadoopFile(outputPath, ImmutableBytesWritable.class, KeyValue.class, HFileOutputFormat2.class, job.getConfiguration());

        // bulk load start

        Table table = connection.getTable(tableName);

        Admin admin = connection.getAdmin();

        LoadIncrementalHFiles load = new LoadIncrementalHFiles(configuration);

        load.doBulkLoad(new Path(outputPath), admin,table,connection.getRegionLocator(tableName));

        jsc.close();

    }

    public static void main(String[] args) {

        try {

            long start = System.currentTimeMillis();

            args = new String[]{"hdfs://master:8020/test/test.txt","hdfs://master:8020/test/hfile/test"};

            run(args);

            long end = System.currentTimeMillis();

            System.out.println("数据导入成功，总计耗时："+(end-start)/1000+"s");

        } catch(Exception e) {

            e.printStackTrace();

        }

    }

}

代码打包，上传到集群执行如下命令：

./spark-submit --master yarn-client --executor-memory 4G --driver-memory 1G --num-executors 100 --executor-cores 4 --total-executor-cores 400 
--conf spark.default.parallelism=1000 --class scala.HbaseBulkLoad /home/hadoop/app/hadoop/data/spark-hbase-test.jar

本次只测试导入了50000条数据，在测试导入15G（1.5亿条左右）数据时，导入速度没有MapReduce快

用spark导入数据到hbase的更多相关文章

批量导入数据到HBase
hbase一般用于大数据的批量分析,所以在很多情况下需要将大量数据从外部导入到hbase中,hbase提供了一种导入数据的方式,主要用于批量导入大量数据,即importtsv工具,用法如下: Us ...
通过phoenix导入数据到hbase出错记录
解决方法1 错误如下 -- ::, [hconnection-0x7b9e01aa-shared--pool11069-t114734] WARN org.apache.hadoop.hbase.ip ...
Hive导入数据到HBase,再与Phoenix映射同步
1. 创建HBase 表 create 'hbase_test','user' 2. 插入数据 put 'hbase_test','111','user:name','jack' put 'hbase ...
importTSV工具导入数据到hbase
1.建立目标表test,确定好列族信息. create'test','info','address' 2.建立文件编写要导入的数据并上传到hdfs上 touch a.csv vi a.csv 数据内容 ...
导入数据到HBase的方式选择
Choosing the Right Import Method If the data is already in an HBase table: To move the data from one ...
使用Sqoop从MySQL导入数据到Hive和HBase 及近期感悟
使用Sqoop从MySQL导入数据到Hive和HBase 及近期感悟 Sqoop 大数据 Hive HBase ETL 使用Sqoop从MySQL导入数据到Hive和HBase 及近期感悟基础环境 ...
Hbase 学习（十一）使用hive往hbase当中导入数据
我们可以有很多方式可以把数据导入到hbase当中,比如说用map-reduce,使用TableOutputFormat这个类,但是这种方式不是最优的方式. Bulk的方式直接生成HFiles,写入到文 ...
教程 | 使用Sqoop从MySQL导入数据到Hive和HBase
基础环境 sqoop:sqoop-1.4.5+cdh5.3.6+78, hive:hive-0.13.1+cdh5.3.6+397, hbase:hbase-0.98.6+cdh5.3.6+115 S ...
Spark实战之读写HBase
1 配置 1.1 开发环境: HBase:hbase-1.0.0-cdh5.4.5.tar.gz Hadoop:hadoop-2.6.0-cdh5.4.5.tar.gz ZooKeeper:zooke ...

随机推荐

JAVA加密技术-----MD5 与SHA 加密
关于JAVA的加密技术有很多很多,这里只介绍加密技术的两种 MD5与 SHA. MD5与SHA是单向加密算法,也就是说加密后不能解密. MD5 ---信息摘要算法,广泛用于加密与解密技术,常用于文件校 ...
ubuntu14.04使用rails连接mysql数据库
rails自带的sqlite3各方面都不错,但是免费版缺少一个致命功能:加密码!虽说第三方有编译好的二进制版的加密版,但咱先不折腾鸟;直接上mysql吧. ubuntu安装mysql非常简单,先不聊; ...
python实现gabor滤波器提取纹理特征提取指静脉纹理特征指静脉切割代码
参考博客:https://blog.csdn.net/xue_wenyuan/article/details/51533953 https://blog.csdn.net/jinshengtao/ar ...
《深入理解java虚拟机》读书笔记1--java内存区域
Java内存管理本文主要介绍Java虚拟机运行时的内存区域是如何划分的.Java对象的创建过程.Java对象的内存布局.Java对象的访问定位一:运行时区域划分主要可以分为以下几个: 程序计数 ...
java队列
"队列"这个单词是英国人说的"排".在英国"排队"的意思就是站到一排当中去.计算机科学中,队列是一种数据结构,有点类似栈,只是在队列中第一个 ...
C++ 延时等待(sleep/timer/wait)
原文链接:http://blog.csdn.net/tangweide/article/details/7063747 (-)使用_sleep()函数 #include <iostream> ...
Android开发之adb无法连接
2017/11/14 21:20 Unable to run 'adb': null 21:20 'E:\AndroidSDK\platform-tools\adb.exe start-server' ...
服务器禁止ping
禁止ping后,不让别人通过域名ping到你的ip, 如果禁用后,你在ping自己的域名会给你返回服务商的IP并提示超时, 这样你就可以减少IP暴露,增加一点安全. 禁止方法: 编辑 /etc/sys ...
DDGScreenShot—截取图片的任意部分
写在前面 DDGScreenShot 库提供了截取任意图片的功能, 支持手势截图,当然,输入任意的区域也可以,下面看看具体的代码代码如下: 方法封装 /** ** 用手势截图(截取图片的任意部分) ...
angular2项目如何使用sass
angular/cli支持使用sass 新建工程: 如果是新建一个angular工程采用sass: ng new My_New_Project --style=sass 这样所有样式的地方都将采用sa ...

用spark导入数据到hbase

用spark导入数据到hbase的更多相关文章

随机推荐

热门专题