datax將同步mysql到hive的脚本和hive ods的full和orc建表語句

m0_37759590

已于 2023-03-11 15:12:44 修改

阅读量383

点赞数

分类专栏： HQL 文章标签： mysql datax hive

于 2023-03-11 11:02:28 首次发布

本文链接：https://blog.csdn.net/m0_37759590/article/details/129460496

版权

HQL 专栏收录该内容

31 篇文章 0 订阅

订阅专栏

datax同步脚本和hive ods的full和orc建表語句

gen_import_config.py

# coding=utf-8
import json
import getopt
import os
import sys
import MySQLdb

#MySQL相关配置，需根据实际情况作出修改
mysql_host = "*****"
mysql_port = "3306"
mysql_user = "*****"
mysql_passwd = "*****"

#HDFS NameNode相关配置，需根据实际情况作出修改
hdfs_nn_host = "hadoop104"
hdfs_nn_port = "8020"

#生成配置文件的目标路径，可根据实际情况作出修改
output_path = "/opt/module/datax/job/import"


def get_connection():
    return MySQLdb.connect(host=mysql_host, port=int(mysql_port), user=mysql_user, passwd=mysql_passwd)


def get_mysql_meta(database, table):
    connection = get_connection()
    cursor = connection.cursor()
    sql = "SELECT COLUMN_NAME,DATA_TYPE from information_schema.COLUMNS WHERE TABLE_SCHEMA=%s AND TABLE_NAME=%s ORDER BY ORDINAL_POSITION"
    cursor.execute(sql, [database, table])
    fetchall = cursor.fetchall()
    cursor.close()
    connection.close()
    return fetchall


def get_mysql_columns(database, table):
    return map(lambda x: x[0], get_mysql_meta(database, table))


def get_hive_columns(database, table):
    def type_mapping(mysql_type):
        mappings = {
            "bigint": "bigint",
            "int": "bigint",
            "smallint": "bigint",
            "tinyint": "bigint",
            "mediumint": "bigint",
            "decimal": "string",
            "double": "double",
            "float": "float",
            "binary": "string",
            "char": "string",
            "varchar": "string",
            "datetime": "string",
            "time": "string",
            "timestamp": "string",
            "date": "string",
            "text": "string"
        }
        return mappings[mysql_type]

    meta = get_mysql_meta(database, table)
    return map(lambda x: {"name": x[0], "type": type_mapping(x[1].lower())}, meta)


def generate_json(source_database, source_table):
    job = {
        "job": {
            "setting": {
                "speed": {
                    "channel": 3
                },
                "errorLimit": {
                    "record": 0,
                    "percentage": 0.02
                }
            },
            "content": [{
                "reader": {
                    "name": "mysqlreader",
                    "parameter": {
                        "username": mysql_user,
                        "password": mysql_passwd,
                        "column": get_mysql_columns(source_database, source_table),
                        "splitPk": "",
                        "connection": [{
                            "table": [source_table],
                            "jdbcUrl": ["jdbc:mysql://" + mysql_host + ":" + mysql_port + "/" + source_database]
                        }]
                    }
                },
                "writer": {
                    "name": "hdfswriter",
                    "parameter": {
                        "defaultFS": "hdfs://" + hdfs_nn_host + ":" + hdfs_nn_port,
                        "fileType": "text",
                        "path": "${targetdir}",
                        "fileName": source_table,
                        "column": get_hive_columns(source_database, source_table),
                        "writeMode": "append",
                        "fieldDelimiter": u'\u0001',
                        "compress": "gzip"
                    }
                },
                "transformer": [

                        {
                          "name": "dx_groovy",
                          "parameter": {
                            "code": "for(int i=0;i<record.getColumnNumber();i++){if(record.getColumn(i).getByteSize()!=0){Column column = record.getColumn(i); def str = column.asString(); def newStr=null; newStr=str.replaceAll(\"[\\r\\n]\",\"\"); record.setColumn(i, new StringColumn(newStr)); };};return record;",
                            "extraPackage":[]
                          }
                        }
                      ]
            }]
        }
    }
    if not os.path.exists(output_path):
        os.makedirs(output_path)
    with open(os.path.join(output_path, ".".join([source_database, source_table, "json"])), "w") as f:
        json.dump(job, f)


def main(args):
    source_database = ""
    source_table = ""

    options, arguments = getopt.getopt(args, '-d:-t:', ['sourcedb=', 'sourcetbl='])
    for opt_name, opt_value in options:
        if opt_name in ('-d', '--sourcedb'):
            source_database = opt_value
        if opt_name in ('-t', '--sourcetbl'):
            source_table = opt_value

    generate_json(source_database, source_table)


if __name__ == '__main__':
    main(sys.argv[1:])

gen_import_config.sh

python ~/bin/gen_import_config.py -d database -t table
python ~/bin/gen_import_config.py -d database -t table
python ~/bin/gen_import_config.py -d database -t table
python ~/bin/gen_import_config.py -d database -t table

#!/bin/bash

DATAX_HOME=/opt/module/datax

# 如果传入日期则do_date等于传入的日期，否则等于前一天日期
if [ -n "$2" ] ;then
    do_date=$2
else
    do_date=`date -d "-1 day" +%F`
fi

#处理目标路径，此处的处理逻辑是，如果目标路径不存在，则创建；若存在，则清空，目的是保证同步任务可重复执行
handle_targetdir() {
  hadoop fs -test -e $1
  if [[ $? -eq 1 ]]; then
    echo "路径$1不存在，正在创建......"
    hadoop fs -mkdir -p $1
  else
    echo "路径$1已经存在"
    fs_count=$(hadoop fs -count $1)
    content_size=$(echo $fs_count | awk '{print $3}')
    if [[ $content_size -eq 0 ]]; then
      echo "路径$1为空"
    else
      echo "路径$1不为空，正在清空......"
      hadoop fs -rm -r -f $1/*
    fi
  fi
}

#数据同步
import_data() {
  datax_config=$1
  target_dir=$2

  handle_targetdir $target_dir
  python $DATAX_HOME/bin/datax.py -p"-Dtargetdir=$target_dir"  $datax_config
}

case $1 in
"a")
  import_data /opt/module/datax/job/import/database.table.json /origin_data/easypm/db/table/$do_date

"all")
  import_data /opt/module/datax/job/import/database.table.json /origin_data/easypm/db/table/$do_date

  ;;
esac

hdfs ods層建表語句 -》 text格式



-- auto-generated definition
DROP TABLE IF EXISTS ods_full;
create external table ods_full
(
    Id              string
) COMMENT '信息表'
    PARTITIONED BY (`dt` STRING)
    ROW FORMAT DELIMITED FIELDS TERMINATED BY '\001'
        lines terminated by '\n'
        NULL DEFINED AS ''
    LOCATION '/warehouse/poc/esay_pm/ods/table';
load data inpath '/origin_data/database/db/table/2023-03-08' into table table partition (dt = '2023-03-08');

hdfs ods層建表語句 -》 orc格式

drop table if exists ods__orc;
create external table ods__orc
(
    Id              string
) COMMENT '信息表'
    PARTITIONED BY (`dt` STRING)
    ROW FORMAT DELIMITED FIELDS TERMINATED BY '\001'
    lines terminated by '\n'
    STORED AS ORC
    LOCATION '/warehouse/poc/database/ods/table'
    TBLPROPERTIES ('orc.compress' = 'snappy');

再使用select將數據導入進去

insert overwrite table ods_tb_todo_full_orc partition (dt = '2023-03-08')
select
       Id,
 
from ods__full;

m0_37759590

关注

0
点赞
踩
0

收藏

觉得还不错? 一键收藏
打赏
0
评论
datax將同步mysql到hive的脚本和hive ods的full和orc建表語句

datax將同步mysql到hive的脚本和hive ods的full和orc建表語句
复制链接

扫一扫

专栏目录

datax將同步mysql到hive的脚本和hive ods的full和orc建表語句

datax同步脚本和hive ods的full和orc建表語句

gen_import_config.sh

“相关推荐”对你有帮助么？