from 
(
select userkey, mos, ct, opa, docid, ch, pagetype
from 
(
from 
(
from
(
select userkey, getmos(mos) as mos, ct, opa, docid, ch, pagetype from client_base_new  where instr(ct, '2015-03-17+') <> 0 and unix_timestamp(ct, 'yyyy-MM-dd+HH:mm:ss') is not null and opa in ('in', 'page', 'end') and dt = '2015-03-17' and userkey <> '#' distribute by userkey sort by userkey, ct) a
select userkey, mos, ct, opa, docid, ch, pagetype where a.userkey <> '' and a.mos <> '' and ct <> '' and opa <> '' and docid <> '' and ch <> '' and a.pagetype <> ''
) m
select userkey, mos, ct, opa, docid, ch, pagetype) b ) c
insert overwrite table client_wen_crumbs PARTITION (dt='2015-03-17')
select transform (userkey, mos, ct, opa, docid, ch, pagetype) USING './client_user_crumbs_tmp2.py' as (userkey, mos, cts, opas, pgids, chs, pgtys) where c.userkey <> '#';

HQL语句报错;语句map阶段正常,reduce阶段大量task被kill掉(不是推测执行);加上limit 10语句可以执行成功。

刚开开始怀疑数据倾斜,

设置

hive.map.aggr=true

hive.groupby.skewindata=true

依然报错。

改变hql写法,重新建表均报错,最后设置:

mapred.reduce.max.attempts = 10(默认是4)

如果map阶段出现这种情况:

mapred.map.max.attempts = 10(默认是4)

执行成功,看来是数据节点出错重试次数朝超过默认值导致的错误。

还有一种修改方式:

mapred.max.map.failures.percent

mapred.max.reduce.failures.percent

也可以执行成功,但是会丢掉部分结果集。

 对于tasktracker被kill掉的原因应该是tasktracker执行任务超时被kill掉了;因此可以设置:mapred.task.timeout属性;

mapred.task.timeout单位是毫秒(千分之一秒)默认600000。

通过以上修改任务执行成功。

最后可以通过设置(单位:字节):

mapred.min.split.size、mapred.max.split.size属性加快任务执行。

以下是报错信息:

MapReduce Total cumulative CPU time: 0 days 2 hours 31 minutes 28 seconds 290 msec
Ended Job = job_201503171745_8465 with errors
Error during job, obtaining debugging information...
Job Tracking URL: http://tongjihadoop1:50030/jobdetails.jsp?jobid=job_201503171745_8465
Examining task ID: task_201503171745_8465_m_000064 (and more) from job job_201503171745_8465
Examining task ID: task_201503171745_8465_m_000053 (and more) from job job_201503171745_8465
Examining task ID: task_201503171745_8465_m_000049 (and more) from job job_201503171745_8465
Examining task ID: task_201503171745_8465_m_000029 (and more) from job job_201503171745_8465
Examining task ID: task_201503171745_8465_m_000034 (and more) from job job_201503171745_8465
Examining task ID: task_201503171745_8465_m_000011 (and more) from job job_201503171745_8465
Examining task ID: task_201503171745_8465_m_000008 (and more) from job job_201503171745_8465
Examining task ID: task_201503171745_8465_m_000008 (and more) from job job_201503171745_8465
Examining task ID: task_201503171745_8465_m_000001 (and more) from job job_201503171745_8465
Examining task ID: task_201503171745_8465_r_000007 (and more) from job job_201503171745_8465
Examining task ID: task_201503171745_8465_r_000008 (and more) from job job_201503171745_8465
Examining task ID: task_201503171745_8465_r_000010 (and more) from job job_201503171745_8465
Examining task ID: task_201503171745_8465_r_000007 (and more) from job job_201503171745_8465
Examining task ID: task_201503171745_8465_r_000012 (and more) from job job_201503171745_8465

Task with the most failures(5):
-----
Task ID:
task_201503171745_8465_r_000007

URL:
http://tongjihadoop1:50030/taskdetails.jsp?jobid=job_201503171745_8465&tipid=task_201503171745_8465_r_000007
-----
Diagnostic Messages for this Task:

FAILED: Execution Error, return code 2 from org.apache.hadoop.hive.ql.exec.MapRedTask
MapReduce Jobs Launched:
Job 0: Map: 63 Reduce: 25 Cumulative CPU: 9088.29 sec HDFS Read: 13138337431 HDFS Write: 1315039273 FAIL
Total MapReduce CPU Time Spent: 0 days 2 hours 31 minutes 28 seconds 290 msec

以下是./client_user_crumbs_tmp2.py:

#!/usr/local/python-2.7.2/bin/python
import sys

# result = ['#uid', '#mos', ['#cts'], ['#opas'], ['#pages'], ['#chs'], ['#pgtys']]
result = ['#', '#', ['#'], ['#'], ['#'], ['#'], ['#']]

def same_pic_page(pgid1, pgid2):
    if pgid1.count('_') == 2 and pgid1.count('_') == pgid2.count('_'):
        return "".join(pgid1.split('_')[:-1]) == "".join(pgid2.split('_')[:-1])
    else:
        return pgid1 == pgid2
    # if pgid1.count('_') == 2 and pgid1.count('_') == pgid2.count('_'):
    #     return "".join(pgid1.split('_')[:-1]) == "".join(pgid2.split('_')[:-1])
    # elif pgid1.count('_') == 2 and pgid2.count('_') == 1:
    #     return "".join(pgid1.split('_')[:-1]) == "".join(pgid2.split('_'))
    # elif pgid2.count('_') == 2 and pgid1.count('_') == 1:
    #     return "".join(pgid2.split('_')[:-1]) == "".join(pgid1.split('_'))
    # elif pgid1.count('_') == 1 and pgid2.count('_') == 1:
    #     return pgid1 == pgid2
    # else:
    #     return False

for line in sys.stdin:
    try:
        tmp = line.split("\t")
        if len(tmp) != 7:
            continue
        items = map(lambda a: a.strip(), tmp)
        uid, mos, ct, opa, docid, ch, pgty = items
        if result[0] != uid:
            if result[0] != '#':
                print "\t".join(map(str, [result[0], result[1], '&'.join(result[2]), '&'.join(result[3]), '&'.join(result[4]), '&'.join(result[5]), '&'.join(result[6])]))
            result = ['#', '#', ['#'], ['#'], ['#'], ['#'], ['#']]
        if result[0] == '#':
            result[0] = uid
            result[1] = mos
            result[2][0] = ct
            result[3][0] = opa
            result[4][0] = docid
            result[5][0] = ch
            result[6][0] = pgty
            continue
        elif pgty == 'pic' and result[6][-1] == 'pic' and same_pic_page(docid, result[4][-1]):
            continue
        if len(result[2]) > 100:
            continue
        result[2].append(ct)
        result[3].append(opa)
        result[4].append(docid)
        result[5].append(ch)
        result[6].append(pgty)
    except:
        continue
View Code

 

posted on 2015-03-24 08:56  闪电战  阅读(2484)  评论(0)    收藏  举报