9（11）6.1.4 DWS层加载数据脚本11

最新推荐文章于 2023-07-12 12:50:58 发布

佑熙

最新推荐文章于 2023-07-12 12:50:58 发布

阅读量154

点赞数

分类专栏：电商数仓2 文章标签：大数据

本文链接：https://blog.csdn.net/weixin_42871374/article/details/105410452

版权

电商数仓2 专栏收录该内容

22 篇文章 1 订阅

订阅专栏

1）在hadoop102的/home/atguigu/bin目录下创建脚本
[atguigu@hadoop102 bin]$ vim dws_uv_log.sh
在脚本中编写如下内容
#!/bin/bash

定义变量方便修改

APP=gmall
hive=/opt/module/hive/bin/hive

如果是输入的日期按照取输入日期；如果没输入日期取当前时间的前一天

if [ -n “$1” ] ;then
do_date=$1
else
do_date=date -d "-1 day" +%F
fi

sql="
set hive.exec.dynamic.partition.mode=nonstrict;

insert overwrite table " $APP".dws_uv_detail_day partition(dt='$ do_date’)
select
mid_id,
concat_ws(’|’, collect_set(user_id)) user_id,
concat_ws(’|’, collect_set(version_code)) version_code,
concat_ws(’|’, collect_set(version_name)) version_name,
concat_ws(’|’, collect_set(lang)) lang,
concat_ws(’|’, collect_set(source)) source,
concat_ws(’|’, collect_set(os)) os,
concat_ws(’|’, collect_set(area)) area,
concat_ws(’|’, collect_set(model)) model,
concat_ws(’|’, collect_set(brand)) brand,
concat_ws(’|’, collect_set(sdk_version)) sdk_version,
concat_ws(’|’, collect_set(gmail)) gmail,
concat_ws(’|’, collect_set(height_width)) height_width,
concat_ws(’|’, collect_set(app_time)) app_time,
concat_ws(’|’, collect_set(network)) network,
concat_ws(’|’, collect_set(lng)) lng,
concat_ws(’|’, collect_set(lat)) lat
from " $APP".dwd_start_log where dt='$ do_date’
group by mid_id;

insert overwrite table “ $APP".dws_uv_detail_wk partition(wk_dt) select mid_id, concat_ws('|', collect_set(user_id)) user_id, concat_ws('|', collect_set(version_code)) version_code, concat_ws('|', collect_set(version_name)) version_name, concat_ws('|', collect_set(lang)) lang, concat_ws('|', collect_set(source)) source, concat_ws('|', collect_set(os)) os, concat_ws('|', collect_set(area)) area, concat_ws('|', collect_set(model)) model, concat_ws('|', collect_set(brand)) brand, concat_ws('|', collect_set(sdk_version)) sdk_version, concat_ws('|', collect_set(gmail)) gmail, concat_ws('|', collect_set(height_width)) height_width, concat_ws('|', collect_set(app_time)) app_time, concat_ws('|', collect_set(network)) network, concat_ws('|', collect_set(lng)) lng, concat_ws('|', collect_set(lat)) lat, date_add(next_day('$ do_date’,‘MO’),-7),
date_add(next_day(‘ $do_date','MO'),-1), concat(date_add( next_day('$ do_date’,‘MO’),-7), ‘_’ , date_add(next_day(' $do_date','MO'),-1) ) from "$ APP”.dws_uv_detail_day
where dt>=date_add(next_day(‘ $do_date','MO'),-7) and dt<=date_add(next_day('$ do_date’,‘MO’),-1)
group by mid_id;

insert overwrite table " $APP".dws_uv_detail_mn partition(mn) select mid_id, concat_ws('|', collect_set(user_id)) user_id, concat_ws('|', collect_set(version_code)) version_code, concat_ws('|', collect_set(version_name)) version_name, concat_ws('|', collect_set(lang))lang, concat_ws('|', collect_set(source)) source, concat_ws('|', collect_set(os)) os, concat_ws('|', collect_set(area)) area, concat_ws('|', collect_set(model)) model, concat_ws('|', collect_set(brand)) brand, concat_ws('|', collect_set(sdk_version)) sdk_version, concat_ws('|', collect_set(gmail)) gmail, concat_ws('|', collect_set(height_width)) height_width, concat_ws('|', collect_set(app_time)) app_time, concat_ws('|', collect_set(network)) network, concat_ws('|', collect_set(lng)) lng, concat_ws('|', collect_set(lat)) lat, date_format('$ do_date’,‘yyyy-MM’)
from " $APP".dws_uv_detail_day where date_format(dt,'yyyy-MM') = date_format('$ do_date’,‘yyyy-MM’)
group by mid_id;
"

$h i v e - e "$ sql"
2）增加脚本执行权限
[atguigu@hadoop102 bin]$ chmod 777 dws_uv_log.sh
3）脚本使用
[atguigu@hadoop102 module]$ dws_uv_log.sh 2019-02-11
4）查询结果
hive (gmall)> select count() from dws_uv_detail_day where dt=‘2019-02-11’;
hive (gmall)> select count() from dws_uv_detail_wk;
hive (gmall)> select count(*) from dws_uv_detail_mn ;
5）脚本执行时间
企业开发中一般在每日凌晨30分~1点