Commit 102f3809 by hejiangming

年度搜索词 新细分市场改为新增词计算逻辑 页面隐藏

parent 8bdae1ea
...@@ -432,8 +432,11 @@ class DwtAbaLast365(object): ...@@ -432,8 +432,11 @@ class DwtAbaLast365(object):
F.concat_ws( F.concat_ws(
",", F.sort_array(F.collect_set(F.split("date_info", "-")[1].cast(IntegerType()))) ",", F.sort_array(F.collect_set(F.split("date_info", "-")[1].cast(IntegerType())))
).alias("total_appear_month"), ).alias("total_appear_month"),
# 是否新细分市场 非平均数算法 12个月都是新出现 表明同比年也是新出现 即 sum=12 表示为1 否则都是0 # ===== 废弃(保留备查):avg 月表同月 is_new_market_segment =====
F.avg("is_new_market_segment").cast(IntegerType()).alias("is_new_market_segment"), # 月表 is_new 只比"本月的去年同月"(M-12),年表又只对今年出现月求 avg;去年对照窗口里
# 没和今年出现月对齐的月(如季节词的去年8月)从没被查到 → 季节词误判为新。
# 改到 handle_label 用"去年整段年表有无该词"独立判定(与 is_first_text 同口径)。
# F.avg("is_new_market_segment").cast(IntegerType()).alias("is_new_market_segment"),
# 同比是否是热搜词 热搜词:最近1月/年中,出现的次数大于80% 如果月热搜词 is_search_text的和>=10 则是热搜词 # 同比是否是热搜词 热搜词:最近1月/年中,出现的次数大于80% 如果月热搜词 is_search_text的和>=10 则是热搜词
F.expr("sum(is_search_text) / 9.6").cast(IntegerType()).alias("is_search_text"), F.expr("sum(is_search_text) / 9.6").cast(IntegerType()).alias("is_search_text"),
# st_attribute_label 12 月合并去重(原来在 handle_month_lastest 只取最新月,最新月该词没出现会丢标签) # st_attribute_label 12 月合并去重(原来在 handle_month_lastest 只取最新月,最新月该词没出现会丢标签)
...@@ -534,6 +537,17 @@ class DwtAbaLast365(object): ...@@ -534,6 +537,17 @@ class DwtAbaLast365(object):
self.df_last_year, on='search_term', how='left' self.df_last_year, on='search_term', how='left'
).fillna({'is_first_text': 1}) ).fillna({'is_first_text': 1})
# 新细分市场:口径="今年窗口有 且 去年整段(去年同期年表 last_year_month)无",与新增词同口径独立判定
# ===== 原逻辑已废弃(见 handle_agg):avg 月表同月 is_new,同月对齐漏看去年非对齐月,季节词误判为新 =====
# df_ly_seg:去年同期年表出现过的词清单,打标 _ly_has=1(去年有该词)
# left join 后 _ly_has 为 null = 去年年表没有该词 = 今年新起的细分市场(1);非 null = 去年已有(0)
# 举例:fall fashion must haves 去年8月在去年年表窗口内 → join 上 → 0(不再误判为新)
df_ly_seg = self.df_last_year.select('search_term').withColumn('_ly_has', F.lit(1))
self.df_base = self.df_base.join(df_ly_seg, on='search_term', how='left').withColumn(
'is_new_market_segment',
F.when(F.col('_ly_has').isNull(), F.lit(1)).otherwise(F.lit(0))
).drop('_ly_has')
# 历史新增词判断 # 历史新增词判断
# ===== 原逻辑(保留备查):靠"是否 join 上 df_history"打 0/1;history 源已换成全历史累加表 dim_st_detail_history ===== # ===== 原逻辑(保留备查):靠"是否 join 上 df_history"打 0/1;history 源已换成全历史累加表 dim_st_detail_history =====
# self.df_base = self.df_base.join( # self.df_base = self.df_base.join(
......
Markdown is supported
0% or
You are about to add 0 people to the discussion. Proceed with caution.
Finish editing this message first!
Please register or to comment