来自官方文档
一、写 python 脚本:
import sys
import datetime
for line in sys.stdin:
line = line.strip()
userid, movieid, rating, unixtime = line.split(' ')
weekday = datetime.datetime.fromtimestamp(float(unixtime)).isoweekday()
print ' '.join([userid, movieid, rating, str(weekday)])
二、添加脚本
add file /opt/datas/xxx.py
三、使用脚本
CREATE TABLE u_data_new (
userid INT,
movieid INT,
rating INT,
weekday INT)
ROW FORMAT DELIMITED
FIELDS TERMINATED BY ' ';
add FILE weekday_mapper.py;
INSERT OVERWRITE TABLE u_data_new
SELECT
TRANSFORM (userid, movieid, rating, unixtime) # 传入参数
USING 'python weekday_mapper.py' # 使用的脚本文件
AS (userid, movieid, rating, weekday) # 输出的字段
FROM u_data;
SELECT weekday, COUNT(*)
FROM u_data_new
GROUP BY weekday;