mirror of
https://github.com/primedigitaltech/market-assistant.git
synced 2026-07-26 10:14:20 +08:00
feat(pipeline): 报告可启用第八章文本挖掘探针替代原 §8.3 关注词与场景
Made-with: Cursor
This commit is contained in:
parent
b19a987a06
commit
31ac2f1ac7
@ -2136,8 +2136,10 @@ def build_competitor_markdown(
|
||||
llm_scenario_groups_section_md: str | None = None,
|
||||
llm_comment_groups_section_md: str | None = None,
|
||||
llm_strategy_opportunities_section_md: str | None = None,
|
||||
chapter8_text_mining_probe_section_md: str | None = None,
|
||||
) -> str:
|
||||
focus_words, scenario_groups, external_rows = resolve_report_tuning(report_config)
|
||||
_ch8_probe_sec = (chapter8_text_mining_probe_section_md or "").strip()
|
||||
sku_header = MERGED_FIELD_TO_CSV_HEADER["sku_id"]
|
||||
title_h = MERGED_FIELD_TO_CSV_HEADER["title"]
|
||||
batch = _run_batch_label(run_dir)
|
||||
@ -2272,7 +2274,11 @@ def build_competitor_markdown(
|
||||
"- **品牌/店铺集中度(第四章)**:有列表全量时按列表行计店铺与品牌占比;无列表导出时按深入 SKU 合并表估算。",
|
||||
"- **评价主题词**:对评价正文做**预设词表子串计数**,非分词主题模型,适合扫方向,**需抽样人工验证**。",
|
||||
"- **用途/场景**:对每条评价独立判断是否命中预设场景词;一条可计入多个场景,统计的是「提及该场景的评价条数」而非用户数。",
|
||||
"- **用户画像(第八章)**:正负面粗判含**口语短语**级摘录;关注词与场景**仅按细类**以**同图左右并列**展示(左为关注词命中次数,右为场景占有效文本 **%**);见 §8.3。",
|
||||
(
|
||||
"- **用户画像(第八章)**:正负面粗判含**口语短语**级摘录;§8.3 为 **jieba + TF-IDF / 共现 / LDA** 文本挖掘探针(与 §8.2 条形图口径不同、互补),可选词云与探针专用大模型归纳。"
|
||||
if _ch8_probe_sec
|
||||
else "- **用户画像(第八章)**:正负面粗判含**口语短语**级摘录;关注词与场景**仅按细类**以**同图左右并列**展示(左为关注词命中次数,右为场景占有效文本 **%**);见 §8.3。"
|
||||
),
|
||||
"- **细类划分(§5~§8)**:**仅**依据合并表 ``detail_category_path``;该列为空或无法解析出可读细类段的 SKU **不参与**竞品矩阵与按细类评价统计(相关评价条亦**不进入**按细类图表)。",
|
||||
"- **检索结果规模**:来自京东 PC 搜索返回的「结果条数」类指标,表示平台侧申报的匹配数量级,**不等于**动销、库存或独立 SKU 数。",
|
||||
"",
|
||||
@ -2726,7 +2732,11 @@ def build_competitor_markdown(
|
||||
"- **细类划分**:与 **§5 竞品矩阵** 相同,**仅**依据 ``detail_category_path`` 解析为「饼干 / 西式糕点 / …」等(规则见 §5 章首说明)。",
|
||||
"- **归因**:每条评价按其 SKU 对应到深入样本,再映射到该 SKU 所属细类;SKU 不在合并表中的评价单独归入说明性分组;**在合并表中但该 SKU 缺 ``detail_category_path`` 或路径无法解析为可读细类的,该评价不进入按细类统计**(与 §5 **同一条排除规则**)。",
|
||||
"- **正负面粗判(§8.2)**:若评价含有效「评分」列则**先按星级**粗分正负与中评,再在对应子集内统计口语短语;无评分时仍按关键词子串;若任务开启 **llm_comment_sentiment**,可附**大模型对抽样原文的主题归因**,与条形图互补。",
|
||||
"- **关注词与使用场景(§8.3)**:对组内评价正文做关注词子串计数(左栏条形图);对每条有效文本独立扫描**本次任务生效的场景词组**(来自报告调参或系统默认),一条可属多场景,右栏为**占该细类有效文本比例 %**(多标签下可相加 **>** 100%)。二者在 **同一张图左右并列**,与 §5 矩阵细类一一对应。",
|
||||
(
|
||||
"- **文本挖掘探针(§8.3)**:本任务已启用 **jieba + sklearn** 的开放词表分析(词频 / TF-IDF / 共现 / LDA,可选词云),与 §8.2 规则词表条形图**口径不同**、**互补**;**不再**输出原「关注词次数 + 场景占比」左右并列条图。"
|
||||
if _ch8_probe_sec
|
||||
else "- **关注词与使用场景(§8.3)**:对组内评价正文做关注词子串计数(左栏条形图);对每条有效文本独立扫描**本次任务生效的场景词组**(来自报告调参或系统默认),一条可属多场景,右栏为**占该细类有效文本比例 %**(多标签下可相加 **>** 100%)。二者在 **同一张图左右并列**,与 §5 矩阵细类一一对应。"
|
||||
),
|
||||
"",
|
||||
_sec82_title,
|
||||
"",
|
||||
@ -2822,6 +2832,18 @@ def build_competitor_markdown(
|
||||
]
|
||||
)
|
||||
lines.append("")
|
||||
if _ch8_probe_sec:
|
||||
lines.extend(
|
||||
[
|
||||
"### 8.3 文本挖掘探针(jieba / TF-IDF / 共现 / LDA)",
|
||||
"",
|
||||
"> **说明**:与 §8.2 口语短语条形图(规则词表)**口径不同**、**互补**;**不**再输出本章原「关注词 + 场景」左右并列条图;插图路径相对于本批次目录下 ``report_assets/``。",
|
||||
"",
|
||||
_ch8_probe_sec,
|
||||
"",
|
||||
]
|
||||
)
|
||||
else:
|
||||
lines.extend(
|
||||
[
|
||||
"### 8.3 关注词与使用场景(按细类)",
|
||||
|
||||
@ -16,6 +16,8 @@
|
||||
python -m pipeline.demos.chapter8_text_mining_probe --run-dir \"...\" --live-llm --llm-chunked
|
||||
|
||||
输出:默认写入 ``<run_dir>/chapter8_text_mining_probe.md``。
|
||||
|
||||
嵌入竞品报告:任务 ``report_config`` 中 ``chapter8_text_mining_probe``: true 时,流水线会生成本稿并调用 ``markdown_embed_body_for_competitor_report`` 写入 ``competitor_analysis.md`` 的 **§8.3**,替代原「关注词 + 场景」条图及对应两段大模型;**§8.2 与「大模型深入解读(主题归因…)」保留**。
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
@ -715,6 +717,34 @@ def build_markdown(
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def markdown_embed_body_for_competitor_report(full_probe_md: str) -> str:
|
||||
"""
|
||||
将独立探针稿转为可嵌入 ``build_competitor_markdown`` 的 **§8.3 正文**(不含 ``### 8.3`` 标题行):
|
||||
从 ``## 8.0 说明`` 起至文末,并把 ``## …`` 降为 ``#### …``,避免与宿主 ``## 八、`` 冲突。
|
||||
"""
|
||||
lines = (full_probe_md or "").splitlines()
|
||||
try:
|
||||
start = next(
|
||||
i
|
||||
for i, ln in enumerate(lines)
|
||||
if ln.strip() == "## 8.0 说明" or ln.strip().startswith("## 8.0 说明")
|
||||
)
|
||||
except StopIteration:
|
||||
return (full_probe_md or "").strip()
|
||||
chunk = lines[start:]
|
||||
out: list[str] = []
|
||||
for ln in chunk:
|
||||
if ln.startswith("## ") and not ln.startswith("###"):
|
||||
out.append("#### " + ln[3:])
|
||||
else:
|
||||
out.append(ln)
|
||||
while out and out[-1].strip() in ("*(完)*", ""):
|
||||
out.pop()
|
||||
while out and not out[-1].strip():
|
||||
out.pop()
|
||||
return "\n".join(out).strip()
|
||||
|
||||
|
||||
def main() -> None:
|
||||
ap = argparse.ArgumentParser(description="第八章文本挖掘探针(独立脚本)")
|
||||
ap.add_argument(
|
||||
|
||||
@ -174,6 +174,10 @@ def get_default_report_config() -> dict[str, Any]:
|
||||
"llm_promo_group_summaries": True,
|
||||
"llm_strategy_opportunities": True,
|
||||
"llm_group_summaries_chunk_by_matrix": True,
|
||||
"chapter8_text_mining_probe": False,
|
||||
"chapter8_text_mining_probe_live_llm": True,
|
||||
"chapter8_text_mining_probe_llm_chunked": True,
|
||||
"chapter8_text_mining_probe_wordcloud": True,
|
||||
"comment_focus_words": list(jcr.COMMENT_FOCUS_WORDS),
|
||||
"comment_scenario_groups": [
|
||||
{"label": lbl, "triggers": list(trs)}
|
||||
@ -458,6 +462,50 @@ def write_competitor_analysis_for_run_dir(
|
||||
"MA_ENABLE_LLM_STRATEGY_OPPORTUNITIES"
|
||||
)
|
||||
|
||||
use_ch8_probe = bool(eff_rc.get("chapter8_text_mining_probe"))
|
||||
chapter8_probe_embed_md = ""
|
||||
ch8_probe_rec: dict[str, Any] = {"schema_version": 1, "attempted": False}
|
||||
if use_ch8_probe:
|
||||
ch8_probe_rec["attempted"] = True
|
||||
try:
|
||||
from ..demos.chapter8_text_mining_probe import (
|
||||
build_markdown as build_ch8_probe_full_md,
|
||||
markdown_embed_body_for_competitor_report,
|
||||
)
|
||||
|
||||
_rc = eff_rc
|
||||
full_probe = build_ch8_probe_full_md(
|
||||
run_dir,
|
||||
min_texts=int(_rc.get("chapter8_probe_min_texts") or 8),
|
||||
lda_topics_n=int(_rc.get("chapter8_probe_lda_topics") or 4),
|
||||
top_k_words=int(_rc.get("chapter8_probe_top_k_words") or 30),
|
||||
cooc_vocab=int(_rc.get("chapter8_probe_cooc_vocab") or 80),
|
||||
cooc_pairs=int(_rc.get("chapter8_probe_cooc_pairs") or 25),
|
||||
live_llm=bool(_rc.get("chapter8_text_mining_probe_live_llm", True)),
|
||||
llm_chunked=bool(
|
||||
_rc.get("chapter8_text_mining_probe_llm_chunked", True)
|
||||
),
|
||||
wordcloud_enabled=bool(
|
||||
_rc.get("chapter8_text_mining_probe_wordcloud", True)
|
||||
),
|
||||
wordcloud_max=int(_rc.get("chapter8_probe_wordcloud_max") or 40),
|
||||
)
|
||||
(run_dir / "chapter8_text_mining_probe.md").write_text(
|
||||
full_probe, encoding="utf-8"
|
||||
)
|
||||
chapter8_probe_embed_md = markdown_embed_body_for_competitor_report(
|
||||
full_probe
|
||||
)
|
||||
ch8_probe_rec["ok"] = True
|
||||
ch8_probe_rec["chars_embed"] = len(chapter8_probe_embed_md)
|
||||
except Exception as e:
|
||||
ch8_probe_rec["ok"] = False
|
||||
ch8_probe_rec["error"] = str(e)
|
||||
|
||||
if use_ch8_probe and chapter8_probe_embed_md:
|
||||
want_sg = False
|
||||
want_cg = False
|
||||
|
||||
chunk_gr = use_chunked_group_summaries_llm(eff_rc)
|
||||
|
||||
if want_mx and not skip_mx and merged_rows:
|
||||
@ -707,6 +755,11 @@ def write_competitor_analysis_for_run_dir(
|
||||
json.dumps(strategy_opp_llm_rec, ensure_ascii=False, indent=2),
|
||||
encoding="utf-8",
|
||||
)
|
||||
if use_ch8_probe:
|
||||
(run_dir / "chapter8_text_mining_probe.json").write_text(
|
||||
json.dumps(ch8_probe_rec, ensure_ascii=False, indent=2),
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
md = jcr.build_competitor_markdown(
|
||||
run_dir=run_dir,
|
||||
@ -723,6 +776,7 @@ def write_competitor_analysis_for_run_dir(
|
||||
llm_scenario_groups_section_md=llm_scenario_gr_md or None,
|
||||
llm_comment_groups_section_md=llm_comment_gr_md or None,
|
||||
llm_strategy_opportunities_section_md=llm_strategy_opp_md or None,
|
||||
chapter8_text_mining_probe_section_md=chapter8_probe_embed_md or None,
|
||||
)
|
||||
|
||||
out_md = run_dir / "competitor_analysis.md"
|
||||
|
||||
@ -24,6 +24,16 @@ _REPORT_CONFIG_ALLOWED_KEYS = frozenset(
|
||||
"llm_comment_group_summaries",
|
||||
"llm_scenario_group_summaries",
|
||||
"llm_group_summaries_chunk_by_matrix",
|
||||
"chapter8_text_mining_probe",
|
||||
"chapter8_text_mining_probe_live_llm",
|
||||
"chapter8_text_mining_probe_llm_chunked",
|
||||
"chapter8_text_mining_probe_wordcloud",
|
||||
"chapter8_probe_min_texts",
|
||||
"chapter8_probe_lda_topics",
|
||||
"chapter8_probe_top_k_words",
|
||||
"chapter8_probe_cooc_vocab",
|
||||
"chapter8_probe_cooc_pairs",
|
||||
"chapter8_probe_wordcloud_max",
|
||||
"comment_focus_words",
|
||||
"comment_scenario_groups",
|
||||
"external_market_table_rows",
|
||||
@ -52,9 +62,24 @@ def validate_report_config_body(value: dict) -> dict:
|
||||
"llm_comment_group_summaries",
|
||||
"llm_scenario_group_summaries",
|
||||
"llm_group_summaries_chunk_by_matrix",
|
||||
"chapter8_text_mining_probe",
|
||||
"chapter8_text_mining_probe_live_llm",
|
||||
"chapter8_text_mining_probe_llm_chunked",
|
||||
"chapter8_text_mining_probe_wordcloud",
|
||||
):
|
||||
if k in value and value[k] is not None and not isinstance(value[k], bool):
|
||||
raise serializers.ValidationError(f"{k} 须为 true 或 false")
|
||||
for k in (
|
||||
"chapter8_probe_min_texts",
|
||||
"chapter8_probe_lda_topics",
|
||||
"chapter8_probe_top_k_words",
|
||||
"chapter8_probe_cooc_vocab",
|
||||
"chapter8_probe_cooc_pairs",
|
||||
"chapter8_probe_wordcloud_max",
|
||||
):
|
||||
if k in value and value[k] is not None:
|
||||
if not isinstance(value[k], int):
|
||||
raise serializers.ValidationError(f"{k} 须为整数")
|
||||
raw = json.dumps(value, ensure_ascii=False)
|
||||
if len(raw) > 120_000:
|
||||
raise serializers.ValidationError("报告配置体积过大")
|
||||
|
||||
Loading…
x
Reference in New Issue
Block a user