Skip to content
Projects
Groups
Snippets
Help
This project
Loading...
Sign in / Register
Toggle navigation
A
Amazon-Selection-Data
Overview
Overview
Details
Activity
Cycle Analytics
Repository
Repository
Files
Commits
Branches
Tags
Contributors
Graph
Compare
Charts
Issues
0
Issues
0
List
Board
Labels
Milestones
Merge Requests
0
Merge Requests
0
CI / CD
CI / CD
Pipelines
Jobs
Schedules
Charts
Wiki
Wiki
Snippets
Snippets
Members
Members
Collapse sidebar
Close sidebar
Activity
Graph
Charts
Create a new issue
Jobs
Commits
Issue Boards
Open sidebar
abel_cjy
Amazon-Selection-Data
Commits
102f3809
Commit
102f3809
authored
Sep 17, 2026
by
hejiangming
Browse files
Options
Browse Files
Download
Email Patches
Plain Diff
年度搜索词 新细分市场改为新增词计算逻辑 页面隐藏
parent
8bdae1ea
Show whitespace changes
Inline
Side-by-side
Showing
1 changed file
with
16 additions
and
2 deletions
+16
-2
dwt_aba_last365.py
Pyspark_job/dwt/dwt_aba_last365.py
+16
-2
No files found.
Pyspark_job/dwt/dwt_aba_last365.py
View file @
102f3809
...
@@ -432,8 +432,11 @@ class DwtAbaLast365(object):
...
@@ -432,8 +432,11 @@ class DwtAbaLast365(object):
F
.
concat_ws
(
F
.
concat_ws
(
","
,
F
.
sort_array
(
F
.
collect_set
(
F
.
split
(
"date_info"
,
"-"
)[
1
]
.
cast
(
IntegerType
())))
","
,
F
.
sort_array
(
F
.
collect_set
(
F
.
split
(
"date_info"
,
"-"
)[
1
]
.
cast
(
IntegerType
())))
)
.
alias
(
"total_appear_month"
),
)
.
alias
(
"total_appear_month"
),
# 是否新细分市场 非平均数算法 12个月都是新出现 表明同比年也是新出现 即 sum=12 表示为1 否则都是0
# ===== 废弃(保留备查):avg 月表同月 is_new_market_segment =====
F
.
avg
(
"is_new_market_segment"
)
.
cast
(
IntegerType
())
.
alias
(
"is_new_market_segment"
),
# 月表 is_new 只比"本月的去年同月"(M-12),年表又只对今年出现月求 avg;去年对照窗口里
# 没和今年出现月对齐的月(如季节词的去年8月)从没被查到 → 季节词误判为新。
# 改到 handle_label 用"去年整段年表有无该词"独立判定(与 is_first_text 同口径)。
# F.avg("is_new_market_segment").cast(IntegerType()).alias("is_new_market_segment"),
# 同比是否是热搜词 热搜词:最近1月/年中,出现的次数大于80% 如果月热搜词 is_search_text的和>=10 则是热搜词
# 同比是否是热搜词 热搜词:最近1月/年中,出现的次数大于80% 如果月热搜词 is_search_text的和>=10 则是热搜词
F
.
expr
(
"sum(is_search_text) / 9.6"
)
.
cast
(
IntegerType
())
.
alias
(
"is_search_text"
),
F
.
expr
(
"sum(is_search_text) / 9.6"
)
.
cast
(
IntegerType
())
.
alias
(
"is_search_text"
),
# st_attribute_label 12 月合并去重(原来在 handle_month_lastest 只取最新月,最新月该词没出现会丢标签)
# st_attribute_label 12 月合并去重(原来在 handle_month_lastest 只取最新月,最新月该词没出现会丢标签)
...
@@ -534,6 +537,17 @@ class DwtAbaLast365(object):
...
@@ -534,6 +537,17 @@ class DwtAbaLast365(object):
self
.
df_last_year
,
on
=
'search_term'
,
how
=
'left'
self
.
df_last_year
,
on
=
'search_term'
,
how
=
'left'
)
.
fillna
({
'is_first_text'
:
1
})
)
.
fillna
({
'is_first_text'
:
1
})
# 新细分市场:口径="今年窗口有 且 去年整段(去年同期年表 last_year_month)无",与新增词同口径独立判定
# ===== 原逻辑已废弃(见 handle_agg):avg 月表同月 is_new,同月对齐漏看去年非对齐月,季节词误判为新 =====
# df_ly_seg:去年同期年表出现过的词清单,打标 _ly_has=1(去年有该词)
# left join 后 _ly_has 为 null = 去年年表没有该词 = 今年新起的细分市场(1);非 null = 去年已有(0)
# 举例:fall fashion must haves 去年8月在去年年表窗口内 → join 上 → 0(不再误判为新)
df_ly_seg
=
self
.
df_last_year
.
select
(
'search_term'
)
.
withColumn
(
'_ly_has'
,
F
.
lit
(
1
))
self
.
df_base
=
self
.
df_base
.
join
(
df_ly_seg
,
on
=
'search_term'
,
how
=
'left'
)
.
withColumn
(
'is_new_market_segment'
,
F
.
when
(
F
.
col
(
'_ly_has'
)
.
isNull
(),
F
.
lit
(
1
))
.
otherwise
(
F
.
lit
(
0
))
)
.
drop
(
'_ly_has'
)
# 历史新增词判断
# 历史新增词判断
# ===== 原逻辑(保留备查):靠"是否 join 上 df_history"打 0/1;history 源已换成全历史累加表 dim_st_detail_history =====
# ===== 原逻辑(保留备查):靠"是否 join 上 df_history"打 0/1;history 源已换成全历史累加表 dim_st_detail_history =====
# self.df_base = self.df_base.join(
# self.df_base = self.df_base.join(
...
...
Write
Preview
Markdown
is supported
0%
Try again
or
attach a new file
Attach a file
Cancel
You are about to add
0
people
to the discussion. Proceed with caution.
Finish editing this message first!
Cancel
Please
register
or
sign in
to comment