mirror of
https://github.com/primedigitaltech/market-assistant.git
synced 2026-07-24 17:21:40 +08:00
feat(pipeline): 报告可启用第八章文本挖掘探针替代原 §8.3 关注词与场景
Made-with: Cursor
This commit is contained in:
parent
b19a987a06
commit
31ac2f1ac7
@ -2136,8 +2136,10 @@ def build_competitor_markdown(
|
|||||||
llm_scenario_groups_section_md: str | None = None,
|
llm_scenario_groups_section_md: str | None = None,
|
||||||
llm_comment_groups_section_md: str | None = None,
|
llm_comment_groups_section_md: str | None = None,
|
||||||
llm_strategy_opportunities_section_md: str | None = None,
|
llm_strategy_opportunities_section_md: str | None = None,
|
||||||
|
chapter8_text_mining_probe_section_md: str | None = None,
|
||||||
) -> str:
|
) -> str:
|
||||||
focus_words, scenario_groups, external_rows = resolve_report_tuning(report_config)
|
focus_words, scenario_groups, external_rows = resolve_report_tuning(report_config)
|
||||||
|
_ch8_probe_sec = (chapter8_text_mining_probe_section_md or "").strip()
|
||||||
sku_header = MERGED_FIELD_TO_CSV_HEADER["sku_id"]
|
sku_header = MERGED_FIELD_TO_CSV_HEADER["sku_id"]
|
||||||
title_h = MERGED_FIELD_TO_CSV_HEADER["title"]
|
title_h = MERGED_FIELD_TO_CSV_HEADER["title"]
|
||||||
batch = _run_batch_label(run_dir)
|
batch = _run_batch_label(run_dir)
|
||||||
@ -2272,7 +2274,11 @@ def build_competitor_markdown(
|
|||||||
"- **品牌/店铺集中度(第四章)**:有列表全量时按列表行计店铺与品牌占比;无列表导出时按深入 SKU 合并表估算。",
|
"- **品牌/店铺集中度(第四章)**:有列表全量时按列表行计店铺与品牌占比;无列表导出时按深入 SKU 合并表估算。",
|
||||||
"- **评价主题词**:对评价正文做**预设词表子串计数**,非分词主题模型,适合扫方向,**需抽样人工验证**。",
|
"- **评价主题词**:对评价正文做**预设词表子串计数**,非分词主题模型,适合扫方向,**需抽样人工验证**。",
|
||||||
"- **用途/场景**:对每条评价独立判断是否命中预设场景词;一条可计入多个场景,统计的是「提及该场景的评价条数」而非用户数。",
|
"- **用途/场景**:对每条评价独立判断是否命中预设场景词;一条可计入多个场景,统计的是「提及该场景的评价条数」而非用户数。",
|
||||||
"- **用户画像(第八章)**:正负面粗判含**口语短语**级摘录;关注词与场景**仅按细类**以**同图左右并列**展示(左为关注词命中次数,右为场景占有效文本 **%**);见 §8.3。",
|
(
|
||||||
|
"- **用户画像(第八章)**:正负面粗判含**口语短语**级摘录;§8.3 为 **jieba + TF-IDF / 共现 / LDA** 文本挖掘探针(与 §8.2 条形图口径不同、互补),可选词云与探针专用大模型归纳。"
|
||||||
|
if _ch8_probe_sec
|
||||||
|
else "- **用户画像(第八章)**:正负面粗判含**口语短语**级摘录;关注词与场景**仅按细类**以**同图左右并列**展示(左为关注词命中次数,右为场景占有效文本 **%**);见 §8.3。"
|
||||||
|
),
|
||||||
"- **细类划分(§5~§8)**:**仅**依据合并表 ``detail_category_path``;该列为空或无法解析出可读细类段的 SKU **不参与**竞品矩阵与按细类评价统计(相关评价条亦**不进入**按细类图表)。",
|
"- **细类划分(§5~§8)**:**仅**依据合并表 ``detail_category_path``;该列为空或无法解析出可读细类段的 SKU **不参与**竞品矩阵与按细类评价统计(相关评价条亦**不进入**按细类图表)。",
|
||||||
"- **检索结果规模**:来自京东 PC 搜索返回的「结果条数」类指标,表示平台侧申报的匹配数量级,**不等于**动销、库存或独立 SKU 数。",
|
"- **检索结果规模**:来自京东 PC 搜索返回的「结果条数」类指标,表示平台侧申报的匹配数量级,**不等于**动销、库存或独立 SKU 数。",
|
||||||
"",
|
"",
|
||||||
@ -2726,7 +2732,11 @@ def build_competitor_markdown(
|
|||||||
"- **细类划分**:与 **§5 竞品矩阵** 相同,**仅**依据 ``detail_category_path`` 解析为「饼干 / 西式糕点 / …」等(规则见 §5 章首说明)。",
|
"- **细类划分**:与 **§5 竞品矩阵** 相同,**仅**依据 ``detail_category_path`` 解析为「饼干 / 西式糕点 / …」等(规则见 §5 章首说明)。",
|
||||||
"- **归因**:每条评价按其 SKU 对应到深入样本,再映射到该 SKU 所属细类;SKU 不在合并表中的评价单独归入说明性分组;**在合并表中但该 SKU 缺 ``detail_category_path`` 或路径无法解析为可读细类的,该评价不进入按细类统计**(与 §5 **同一条排除规则**)。",
|
"- **归因**:每条评价按其 SKU 对应到深入样本,再映射到该 SKU 所属细类;SKU 不在合并表中的评价单独归入说明性分组;**在合并表中但该 SKU 缺 ``detail_category_path`` 或路径无法解析为可读细类的,该评价不进入按细类统计**(与 §5 **同一条排除规则**)。",
|
||||||
"- **正负面粗判(§8.2)**:若评价含有效「评分」列则**先按星级**粗分正负与中评,再在对应子集内统计口语短语;无评分时仍按关键词子串;若任务开启 **llm_comment_sentiment**,可附**大模型对抽样原文的主题归因**,与条形图互补。",
|
"- **正负面粗判(§8.2)**:若评价含有效「评分」列则**先按星级**粗分正负与中评,再在对应子集内统计口语短语;无评分时仍按关键词子串;若任务开启 **llm_comment_sentiment**,可附**大模型对抽样原文的主题归因**,与条形图互补。",
|
||||||
"- **关注词与使用场景(§8.3)**:对组内评价正文做关注词子串计数(左栏条形图);对每条有效文本独立扫描**本次任务生效的场景词组**(来自报告调参或系统默认),一条可属多场景,右栏为**占该细类有效文本比例 %**(多标签下可相加 **>** 100%)。二者在 **同一张图左右并列**,与 §5 矩阵细类一一对应。",
|
(
|
||||||
|
"- **文本挖掘探针(§8.3)**:本任务已启用 **jieba + sklearn** 的开放词表分析(词频 / TF-IDF / 共现 / LDA,可选词云),与 §8.2 规则词表条形图**口径不同**、**互补**;**不再**输出原「关注词次数 + 场景占比」左右并列条图。"
|
||||||
|
if _ch8_probe_sec
|
||||||
|
else "- **关注词与使用场景(§8.3)**:对组内评价正文做关注词子串计数(左栏条形图);对每条有效文本独立扫描**本次任务生效的场景词组**(来自报告调参或系统默认),一条可属多场景,右栏为**占该细类有效文本比例 %**(多标签下可相加 **>** 100%)。二者在 **同一张图左右并列**,与 §5 矩阵细类一一对应。"
|
||||||
|
),
|
||||||
"",
|
"",
|
||||||
_sec82_title,
|
_sec82_title,
|
||||||
"",
|
"",
|
||||||
@ -2822,87 +2832,99 @@ def build_competitor_markdown(
|
|||||||
]
|
]
|
||||||
)
|
)
|
||||||
lines.append("")
|
lines.append("")
|
||||||
lines.extend(
|
if _ch8_probe_sec:
|
||||||
[
|
lines.extend(
|
||||||
"### 8.3 关注词与使用场景(按细类)",
|
[
|
||||||
"",
|
"### 8.3 文本挖掘探针(jieba / TF-IDF / 共现 / LDA)",
|
||||||
"每细类一张**左右并列图**(与 ``report_assets/chart_focus_and_scenarios_bar__*.png`` 同源):"
|
"",
|
||||||
"**左**为配置关注词子串命中次数(同一评价可出现多次,为次数而非去重条数);"
|
"> **说明**:与 §8.2 口语短语条形图(规则词表)**口径不同**、**互补**;**不**再输出本章原「关注词 + 场景」左右并列条图;插图路径相对于本批次目录下 ``report_assets/``。",
|
||||||
"**右**为预设场景词组命中占该细类有效文本比例 %(一条可属多场景;多柱比例可相加 **>** 100%)。"
|
"",
|
||||||
"统计均基于评价正文(或兜底预览)子串规则,**不等于**购买动机调研结论。",
|
_ch8_probe_sec,
|
||||||
"",
|
"",
|
||||||
]
|
]
|
||||||
)
|
)
|
||||||
if not feedback_groups:
|
|
||||||
lines.append("*无评价数据可归组。*")
|
|
||||||
lines.append("")
|
|
||||||
else:
|
else:
|
||||||
for gi, (gname, cr_g, texts_g) in enumerate(feedback_groups):
|
lines.extend(
|
||||||
n_flat = len(cr_g)
|
[
|
||||||
lines.append(f"#### {gname}")
|
"### 8.3 关注词与使用场景(按细类)",
|
||||||
|
"",
|
||||||
|
"每细类一张**左右并列图**(与 ``report_assets/chart_focus_and_scenarios_bar__*.png`` 同源):"
|
||||||
|
"**左**为配置关注词子串命中次数(同一评价可出现多次,为次数而非去重条数);"
|
||||||
|
"**右**为预设场景词组命中占该细类有效文本比例 %(一条可属多场景;多柱比例可相加 **>** 100%)。"
|
||||||
|
"统计均基于评价正文(或兜底预览)子串规则,**不等于**购买动机调研结论。",
|
||||||
|
"",
|
||||||
|
]
|
||||||
|
)
|
||||||
|
if not feedback_groups:
|
||||||
|
lines.append("*无评价数据可归组。*")
|
||||||
lines.append("")
|
lines.append("")
|
||||||
lines.append(
|
else:
|
||||||
f"- **本细类逐条评价**:{n_flat} 条;**用于统计的有效文本条数**:{len(texts_g)}。"
|
for gi, (gname, cr_g, texts_g) in enumerate(feedback_groups):
|
||||||
)
|
n_flat = len(cr_g)
|
||||||
lines.append("")
|
lines.append(f"#### {gname}")
|
||||||
hits_g = _group_keyword_hits(cr_g, texts_g, focus_words=focus_words)
|
|
||||||
scen_g, scen_ng = _comment_scenario_counts(texts_g, scenario_groups)
|
|
||||||
has_focus = any(n > 0 for n in hits_g.values()) if hits_g else False
|
|
||||||
has_scen = scen_ng > 0 and any(n > 0 for n in scen_g.values())
|
|
||||||
if scen_ng <= 0:
|
|
||||||
lines.append("*该细类下无可用评价正文。*")
|
|
||||||
lines.append("")
|
lines.append("")
|
||||||
continue
|
lines.append(
|
||||||
if has_focus or has_scen:
|
f"- **本细类逐条评价**:{n_flat} 条;**用于统计的有效文本条数**:{len(texts_g)}。"
|
||||||
cap = (
|
|
||||||
f"「{_md_cell(gname, 24)}」细类 · 关注词与使用场景(左:关注词命中次数;右:场景占有效文本 %;"
|
|
||||||
f"有效文本 **{scen_ng}** 条)"
|
|
||||||
)
|
)
|
||||||
lines.extend(
|
|
||||||
_embed_chart(
|
|
||||||
run_dir,
|
|
||||||
_focus_scenario_combo_bar_filename(gname, gi),
|
|
||||||
cap,
|
|
||||||
)
|
|
||||||
)
|
|
||||||
else:
|
|
||||||
lines.append("*该细类无关注词命中且未命中预设场景词组。*")
|
|
||||||
lines.append("")
|
lines.append("")
|
||||||
if has_scen:
|
hits_g = _group_keyword_hits(cr_g, texts_g, focus_words=focus_words)
|
||||||
for para in _scenario_summary_bullets(scen_g, scen_ng):
|
scen_g, scen_ng = _comment_scenario_counts(texts_g, scenario_groups)
|
||||||
lines.append(para)
|
has_focus = any(n > 0 for n in hits_g.values()) if hits_g else False
|
||||||
|
has_scen = scen_ng > 0 and any(n > 0 for n in scen_g.values())
|
||||||
|
if scen_ng <= 0:
|
||||||
|
lines.append("*该细类下无可用评价正文。*")
|
||||||
|
lines.append("")
|
||||||
|
continue
|
||||||
|
if has_focus or has_scen:
|
||||||
|
cap = (
|
||||||
|
f"「{_md_cell(gname, 24)}」细类 · 关注词与使用场景(左:关注词命中次数;右:场景占有效文本 %;"
|
||||||
|
f"有效文本 **{scen_ng}** 条)"
|
||||||
|
)
|
||||||
|
lines.extend(
|
||||||
|
_embed_chart(
|
||||||
|
run_dir,
|
||||||
|
_focus_scenario_combo_bar_filename(gname, gi),
|
||||||
|
cap,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
lines.append("*该细类无关注词命中且未命中预设场景词组。*")
|
||||||
|
lines.append("")
|
||||||
|
if has_scen:
|
||||||
|
for para in _scenario_summary_bullets(scen_g, scen_ng):
|
||||||
|
lines.append(para)
|
||||||
|
lines.append("")
|
||||||
|
elif scen_ng > 0:
|
||||||
|
lines.append("*未命中预设场景词组。*")
|
||||||
lines.append("")
|
lines.append("")
|
||||||
elif scen_ng > 0:
|
|
||||||
lines.append("*未命中预设场景词组。*")
|
|
||||||
lines.append("")
|
|
||||||
|
|
||||||
_llm_sg = (llm_scenario_groups_section_md or "").strip()
|
_llm_sg = (llm_scenario_groups_section_md or "").strip()
|
||||||
if _llm_sg:
|
if _llm_sg:
|
||||||
lines.extend(
|
lines.extend(
|
||||||
[
|
[
|
||||||
"",
|
"",
|
||||||
"#### 使用场景要点归纳(大模型,与 §8.3 右栏图表互补)",
|
"#### 使用场景要点归纳(大模型,与 §8.3 右栏图表互补)",
|
||||||
"",
|
"",
|
||||||
"> **说明**:与 §8.3 **相同**的预设场景词组与子串命中规则;**各场景条数与占比以正文图右栏为准**。",
|
"> **说明**:与 §8.3 **相同**的预设场景词组与子串命中规则;**各场景条数与占比以正文图右栏为准**。",
|
||||||
"",
|
"",
|
||||||
_llm_sg,
|
_llm_sg,
|
||||||
"",
|
"",
|
||||||
]
|
]
|
||||||
)
|
)
|
||||||
|
|
||||||
_llm_cg = (llm_comment_groups_section_md or "").strip()
|
_llm_cg = (llm_comment_groups_section_md or "").strip()
|
||||||
if _llm_cg:
|
if _llm_cg:
|
||||||
lines.extend(
|
lines.extend(
|
||||||
[
|
[
|
||||||
"",
|
"",
|
||||||
"#### 细类评价与关注词要点归纳(大模型,与 §8.3 左栏图表互补)",
|
"#### 细类评价与关注词要点归纳(大模型,与 §8.3 左栏图表互补)",
|
||||||
"",
|
"",
|
||||||
"> **说明**:归纳各细类反馈主题与配置关注词命中;**次数与 §8.3 图左栏以正文为准**。",
|
"> **说明**:归纳各细类反馈主题与配置关注词命中;**次数与 §8.3 图左栏以正文为准**。",
|
||||||
"",
|
"",
|
||||||
_llm_cg,
|
_llm_cg,
|
||||||
"",
|
"",
|
||||||
]
|
]
|
||||||
)
|
)
|
||||||
|
|
||||||
lines.extend(["---", "", "## 九、策略与机会提示(假设清单,待验证)", ""])
|
lines.extend(["---", "", "## 九、策略与机会提示(假设清单,待验证)", ""])
|
||||||
_llm_st = (llm_strategy_opportunities_section_md or "").strip()
|
_llm_st = (llm_strategy_opportunities_section_md or "").strip()
|
||||||
|
|||||||
@ -16,6 +16,8 @@
|
|||||||
python -m pipeline.demos.chapter8_text_mining_probe --run-dir \"...\" --live-llm --llm-chunked
|
python -m pipeline.demos.chapter8_text_mining_probe --run-dir \"...\" --live-llm --llm-chunked
|
||||||
|
|
||||||
输出:默认写入 ``<run_dir>/chapter8_text_mining_probe.md``。
|
输出:默认写入 ``<run_dir>/chapter8_text_mining_probe.md``。
|
||||||
|
|
||||||
|
嵌入竞品报告:任务 ``report_config`` 中 ``chapter8_text_mining_probe``: true 时,流水线会生成本稿并调用 ``markdown_embed_body_for_competitor_report`` 写入 ``competitor_analysis.md`` 的 **§8.3**,替代原「关注词 + 场景」条图及对应两段大模型;**§8.2 与「大模型深入解读(主题归因…)」保留**。
|
||||||
"""
|
"""
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
@ -715,6 +717,34 @@ def build_markdown(
|
|||||||
return "\n".join(lines)
|
return "\n".join(lines)
|
||||||
|
|
||||||
|
|
||||||
|
def markdown_embed_body_for_competitor_report(full_probe_md: str) -> str:
|
||||||
|
"""
|
||||||
|
将独立探针稿转为可嵌入 ``build_competitor_markdown`` 的 **§8.3 正文**(不含 ``### 8.3`` 标题行):
|
||||||
|
从 ``## 8.0 说明`` 起至文末,并把 ``## …`` 降为 ``#### …``,避免与宿主 ``## 八、`` 冲突。
|
||||||
|
"""
|
||||||
|
lines = (full_probe_md or "").splitlines()
|
||||||
|
try:
|
||||||
|
start = next(
|
||||||
|
i
|
||||||
|
for i, ln in enumerate(lines)
|
||||||
|
if ln.strip() == "## 8.0 说明" or ln.strip().startswith("## 8.0 说明")
|
||||||
|
)
|
||||||
|
except StopIteration:
|
||||||
|
return (full_probe_md or "").strip()
|
||||||
|
chunk = lines[start:]
|
||||||
|
out: list[str] = []
|
||||||
|
for ln in chunk:
|
||||||
|
if ln.startswith("## ") and not ln.startswith("###"):
|
||||||
|
out.append("#### " + ln[3:])
|
||||||
|
else:
|
||||||
|
out.append(ln)
|
||||||
|
while out and out[-1].strip() in ("*(完)*", ""):
|
||||||
|
out.pop()
|
||||||
|
while out and not out[-1].strip():
|
||||||
|
out.pop()
|
||||||
|
return "\n".join(out).strip()
|
||||||
|
|
||||||
|
|
||||||
def main() -> None:
|
def main() -> None:
|
||||||
ap = argparse.ArgumentParser(description="第八章文本挖掘探针(独立脚本)")
|
ap = argparse.ArgumentParser(description="第八章文本挖掘探针(独立脚本)")
|
||||||
ap.add_argument(
|
ap.add_argument(
|
||||||
|
|||||||
@ -174,6 +174,10 @@ def get_default_report_config() -> dict[str, Any]:
|
|||||||
"llm_promo_group_summaries": True,
|
"llm_promo_group_summaries": True,
|
||||||
"llm_strategy_opportunities": True,
|
"llm_strategy_opportunities": True,
|
||||||
"llm_group_summaries_chunk_by_matrix": True,
|
"llm_group_summaries_chunk_by_matrix": True,
|
||||||
|
"chapter8_text_mining_probe": False,
|
||||||
|
"chapter8_text_mining_probe_live_llm": True,
|
||||||
|
"chapter8_text_mining_probe_llm_chunked": True,
|
||||||
|
"chapter8_text_mining_probe_wordcloud": True,
|
||||||
"comment_focus_words": list(jcr.COMMENT_FOCUS_WORDS),
|
"comment_focus_words": list(jcr.COMMENT_FOCUS_WORDS),
|
||||||
"comment_scenario_groups": [
|
"comment_scenario_groups": [
|
||||||
{"label": lbl, "triggers": list(trs)}
|
{"label": lbl, "triggers": list(trs)}
|
||||||
@ -458,6 +462,50 @@ def write_competitor_analysis_for_run_dir(
|
|||||||
"MA_ENABLE_LLM_STRATEGY_OPPORTUNITIES"
|
"MA_ENABLE_LLM_STRATEGY_OPPORTUNITIES"
|
||||||
)
|
)
|
||||||
|
|
||||||
|
use_ch8_probe = bool(eff_rc.get("chapter8_text_mining_probe"))
|
||||||
|
chapter8_probe_embed_md = ""
|
||||||
|
ch8_probe_rec: dict[str, Any] = {"schema_version": 1, "attempted": False}
|
||||||
|
if use_ch8_probe:
|
||||||
|
ch8_probe_rec["attempted"] = True
|
||||||
|
try:
|
||||||
|
from ..demos.chapter8_text_mining_probe import (
|
||||||
|
build_markdown as build_ch8_probe_full_md,
|
||||||
|
markdown_embed_body_for_competitor_report,
|
||||||
|
)
|
||||||
|
|
||||||
|
_rc = eff_rc
|
||||||
|
full_probe = build_ch8_probe_full_md(
|
||||||
|
run_dir,
|
||||||
|
min_texts=int(_rc.get("chapter8_probe_min_texts") or 8),
|
||||||
|
lda_topics_n=int(_rc.get("chapter8_probe_lda_topics") or 4),
|
||||||
|
top_k_words=int(_rc.get("chapter8_probe_top_k_words") or 30),
|
||||||
|
cooc_vocab=int(_rc.get("chapter8_probe_cooc_vocab") or 80),
|
||||||
|
cooc_pairs=int(_rc.get("chapter8_probe_cooc_pairs") or 25),
|
||||||
|
live_llm=bool(_rc.get("chapter8_text_mining_probe_live_llm", True)),
|
||||||
|
llm_chunked=bool(
|
||||||
|
_rc.get("chapter8_text_mining_probe_llm_chunked", True)
|
||||||
|
),
|
||||||
|
wordcloud_enabled=bool(
|
||||||
|
_rc.get("chapter8_text_mining_probe_wordcloud", True)
|
||||||
|
),
|
||||||
|
wordcloud_max=int(_rc.get("chapter8_probe_wordcloud_max") or 40),
|
||||||
|
)
|
||||||
|
(run_dir / "chapter8_text_mining_probe.md").write_text(
|
||||||
|
full_probe, encoding="utf-8"
|
||||||
|
)
|
||||||
|
chapter8_probe_embed_md = markdown_embed_body_for_competitor_report(
|
||||||
|
full_probe
|
||||||
|
)
|
||||||
|
ch8_probe_rec["ok"] = True
|
||||||
|
ch8_probe_rec["chars_embed"] = len(chapter8_probe_embed_md)
|
||||||
|
except Exception as e:
|
||||||
|
ch8_probe_rec["ok"] = False
|
||||||
|
ch8_probe_rec["error"] = str(e)
|
||||||
|
|
||||||
|
if use_ch8_probe and chapter8_probe_embed_md:
|
||||||
|
want_sg = False
|
||||||
|
want_cg = False
|
||||||
|
|
||||||
chunk_gr = use_chunked_group_summaries_llm(eff_rc)
|
chunk_gr = use_chunked_group_summaries_llm(eff_rc)
|
||||||
|
|
||||||
if want_mx and not skip_mx and merged_rows:
|
if want_mx and not skip_mx and merged_rows:
|
||||||
@ -707,6 +755,11 @@ def write_competitor_analysis_for_run_dir(
|
|||||||
json.dumps(strategy_opp_llm_rec, ensure_ascii=False, indent=2),
|
json.dumps(strategy_opp_llm_rec, ensure_ascii=False, indent=2),
|
||||||
encoding="utf-8",
|
encoding="utf-8",
|
||||||
)
|
)
|
||||||
|
if use_ch8_probe:
|
||||||
|
(run_dir / "chapter8_text_mining_probe.json").write_text(
|
||||||
|
json.dumps(ch8_probe_rec, ensure_ascii=False, indent=2),
|
||||||
|
encoding="utf-8",
|
||||||
|
)
|
||||||
|
|
||||||
md = jcr.build_competitor_markdown(
|
md = jcr.build_competitor_markdown(
|
||||||
run_dir=run_dir,
|
run_dir=run_dir,
|
||||||
@ -723,6 +776,7 @@ def write_competitor_analysis_for_run_dir(
|
|||||||
llm_scenario_groups_section_md=llm_scenario_gr_md or None,
|
llm_scenario_groups_section_md=llm_scenario_gr_md or None,
|
||||||
llm_comment_groups_section_md=llm_comment_gr_md or None,
|
llm_comment_groups_section_md=llm_comment_gr_md or None,
|
||||||
llm_strategy_opportunities_section_md=llm_strategy_opp_md or None,
|
llm_strategy_opportunities_section_md=llm_strategy_opp_md or None,
|
||||||
|
chapter8_text_mining_probe_section_md=chapter8_probe_embed_md or None,
|
||||||
)
|
)
|
||||||
|
|
||||||
out_md = run_dir / "competitor_analysis.md"
|
out_md = run_dir / "competitor_analysis.md"
|
||||||
|
|||||||
@ -24,6 +24,16 @@ _REPORT_CONFIG_ALLOWED_KEYS = frozenset(
|
|||||||
"llm_comment_group_summaries",
|
"llm_comment_group_summaries",
|
||||||
"llm_scenario_group_summaries",
|
"llm_scenario_group_summaries",
|
||||||
"llm_group_summaries_chunk_by_matrix",
|
"llm_group_summaries_chunk_by_matrix",
|
||||||
|
"chapter8_text_mining_probe",
|
||||||
|
"chapter8_text_mining_probe_live_llm",
|
||||||
|
"chapter8_text_mining_probe_llm_chunked",
|
||||||
|
"chapter8_text_mining_probe_wordcloud",
|
||||||
|
"chapter8_probe_min_texts",
|
||||||
|
"chapter8_probe_lda_topics",
|
||||||
|
"chapter8_probe_top_k_words",
|
||||||
|
"chapter8_probe_cooc_vocab",
|
||||||
|
"chapter8_probe_cooc_pairs",
|
||||||
|
"chapter8_probe_wordcloud_max",
|
||||||
"comment_focus_words",
|
"comment_focus_words",
|
||||||
"comment_scenario_groups",
|
"comment_scenario_groups",
|
||||||
"external_market_table_rows",
|
"external_market_table_rows",
|
||||||
@ -52,9 +62,24 @@ def validate_report_config_body(value: dict) -> dict:
|
|||||||
"llm_comment_group_summaries",
|
"llm_comment_group_summaries",
|
||||||
"llm_scenario_group_summaries",
|
"llm_scenario_group_summaries",
|
||||||
"llm_group_summaries_chunk_by_matrix",
|
"llm_group_summaries_chunk_by_matrix",
|
||||||
|
"chapter8_text_mining_probe",
|
||||||
|
"chapter8_text_mining_probe_live_llm",
|
||||||
|
"chapter8_text_mining_probe_llm_chunked",
|
||||||
|
"chapter8_text_mining_probe_wordcloud",
|
||||||
):
|
):
|
||||||
if k in value and value[k] is not None and not isinstance(value[k], bool):
|
if k in value and value[k] is not None and not isinstance(value[k], bool):
|
||||||
raise serializers.ValidationError(f"{k} 须为 true 或 false")
|
raise serializers.ValidationError(f"{k} 须为 true 或 false")
|
||||||
|
for k in (
|
||||||
|
"chapter8_probe_min_texts",
|
||||||
|
"chapter8_probe_lda_topics",
|
||||||
|
"chapter8_probe_top_k_words",
|
||||||
|
"chapter8_probe_cooc_vocab",
|
||||||
|
"chapter8_probe_cooc_pairs",
|
||||||
|
"chapter8_probe_wordcloud_max",
|
||||||
|
):
|
||||||
|
if k in value and value[k] is not None:
|
||||||
|
if not isinstance(value[k], int):
|
||||||
|
raise serializers.ValidationError(f"{k} 须为整数")
|
||||||
raw = json.dumps(value, ensure_ascii=False)
|
raw = json.dumps(value, ensure_ascii=False)
|
||||||
if len(raw) > 120_000:
|
if len(raw) > 120_000:
|
||||||
raise serializers.ValidationError("报告配置体积过大")
|
raise serializers.ValidationError("报告配置体积过大")
|
||||||
|
|||||||
Loading…
x
Reference in New Issue
Block a user