hive案例

ods

create table house_ods_table(

region string,

subway_station string,

type string,

area int,

floor_level string,

total_price int,

unit_price int,

distance string)

row format delimited fields terminated by '\t'

location '/hive';

load data local inpath '/opt/datas/house.txt'

overwrite into table house_ods_table;

dwd

create table house_dwd_table (

subway_station string,

type string,

area double,

floor_level string,

total_floor int,

total_price int,

unit_price int,

distance string

)

partitioned by (region string)

row format delimited

fields terminated by '\t';

set hive.exec.dynamic.partition=true;

set hive.exec.dynamic.partition.mode=nonstrict;

INSERT INTO table house_dwd_table PARTITION(region)

SELECT

subway_station,

type,

area,

substring_index(floor_level,'(',1) as floor_level, substring_index(substring_index(floor_level,'共',-1),'层',1) as total_floor,

total_price,

unit_price,

distance,

region

FROM house_ods_database.house_ods_table;

dws

create table priceavg_dws_table(

priceavg double,

type string,

area double,

floor_leval string,

distance string,

group_type string)

row format delimited fields terminated by'\t';

insert into table priceavg_dws_table

select avg(unit_price) priceavg,type,-1 as area,'-1' as floor_level,'-1' as distance,'1' as group_type

from house_dwd_database.house_dwd_table

where region='CPQ'

group by type;

dws

create table salenum_dws_table(

salenum double,

type string,

area double,

floor_leval string,

distance string,

group_type string)

row format delimited fields terminated by'\t';

相关推荐
卷毛迷你猪1 天前
快速实验篇(A11)数据集成与多维分析:从单实验产出到跨实验宽表
hive·hadoop
西木莉1 天前
数据仓库概述
数据仓库
卷毛迷你猪2 天前
快速实验篇(A10)短序列上 SPI 的失效机制
hadoop
Francek Chen2 天前
【大数据处理与分析】数据仓库Hive:04 数据仓库Hive概述
大数据·数据仓库·hive·hadoop·分布式
Gl�ria2 天前
Hadoop/YARN 集群缩容:下线DN节点
大数据·hadoop·分布式
计算机源码社3 天前
基于大数据技术的台北市住宅价格影响因素挖掘与可视化分析-基于Python与Hadoop的台北市住宅价格数据仓库构建与可视化
大数据·hadoop·python·数据分析·spark·毕业设计·数据可视化
计算机源码社3 天前
基于Hadoop+Spark的乳腺癌病理数据可视化分析系统 基于K-Means聚类与PCA降维的乳腺癌形态特征分析系统
大数据·hadoop·python·数据分析·spark·毕业设计·数据可视化
Moshow郑锴3 天前
从“水库”到“直饮水站”:重新理解 Data Mart 与 Data Lake、Data Lakehouse、Data Warehouse 的区别
数据仓库·data·湖仓一体
躺柒4 天前
读数据架构知识体系指南01关系数据仓库(上)
数据仓库·架构·数据分析·spark·数据湖·关系数据库·企业数据仓库
卷毛迷你猪4 天前
快速实验篇(A9-2)Python vs MapReduce:小批量任务的工具选择
大数据·hadoop