From a8210ec9335ce8feee135477ffb26a84300d1ddc Mon Sep 17 00:00:00 2001 From: hub-gif <2487812171@qq.com> Date: Mon, 13 Apr 2026 11:06:35 +0800 Subject: [PATCH] =?UTF-8?q?feat(=E6=8A=A5=E5=91=8A):=20=E5=90=84=E7=AB=A0?= =?UTF-8?q?=E5=A4=A7=E6=A8=A1=E5=9E=8B=E8=A1=94=E6=8E=A5=E5=88=86=E6=9E=90?= =?UTF-8?q?=E4=B8=8E=20report=5Fconfig=20=E5=BC=80=E5=85=B3?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Made-with: Cursor --- .env.example | 3 + .../jd_pc_search/jd_competitor_report.py | 1 + backend/pipeline/jd_runner.py | 63 ++++++++++ backend/pipeline/llm_generate.py | 108 ++++++++++++++++++ backend/pipeline/serializers.py | 5 + .../src/composables/useReportConfigForm.js | 12 ++ frontend/src/views/jd/JdAnalysisBuildView.vue | 14 ++- 7 files changed, 205 insertions(+), 1 deletion(-) diff --git a/.env.example b/.env.example index b19a7fa..19153bb 100644 --- a/.env.example +++ b/.env.example @@ -33,3 +33,6 @@ CSRF_TRUSTED_ORIGINS=http://localhost:5173,http://127.0.0.1:5173 # MA_SKIP_LLM_KEYWORD_SUGGEST=1 # MA_ENABLE_LLM_COMMENT_SENTIMENT=1 # MA_SKIP_LLM_COMMENT_SENTIMENT=1 +# 各章标题后插入大模型「衔接分析」(任务 report_config.llm_section_bridges 或本项设为 1) +# MA_ENABLE_LLM_SECTION_BRIDGES=1 +# MA_SKIP_LLM_SECTION_BRIDGES=1 diff --git a/backend/crawler_copy/jd_pc_search/jd_competitor_report.py b/backend/crawler_copy/jd_pc_search/jd_competitor_report.py index b3a68eb..ec667a2 100644 --- a/backend/crawler_copy/jd_pc_search/jd_competitor_report.py +++ b/backend/crawler_copy/jd_pc_search/jd_competitor_report.py @@ -1544,6 +1544,7 @@ def build_competitor_markdown( "- **评价主题词**:对评价正文做**预设词表子串计数**,非分词主题模型,适合扫方向,**需抽样人工验证**。", "- **用途/场景**:对每条评价独立判断是否命中预设场景词;一条可计入多个场景,统计的是「提及该场景的评价条数」而非用户数。", "- **用户画像(第八章)**:正负面粗判含**口语短语**级摘录;关注词与场景**仅按细类**以条形图展示(场景图为**占该细类有效文本比例 %**);见 §8.3~8.4。", + "- **各章衔接(可选)**:若任务配置 ``llm_section_bridges``(或部署侧环境变量启用),则在「## 一」至「## 九」各章二级标题后插入大模型撰写的**衔接分析**段落,便于阅读过渡;**定量结论仍以正文表格与摘要 JSON 为准**。", "- **检索结果规模**:来自京东 PC 搜索返回的「结果条数」类指标,表示平台侧申报的匹配数量级,**不等于**动销、库存或独立 SKU 数。", "", "### 1.4 主要局限", diff --git a/backend/pipeline/jd_runner.py b/backend/pipeline/jd_runner.py index f509d3f..e5cfb94 100644 --- a/backend/pipeline/jd_runner.py +++ b/backend/pipeline/jd_runner.py @@ -69,6 +69,25 @@ def merge_llm_report_with_rules_charts(llm_md: str, rules_md: str) -> str: return merge_llm_supplement_with_rules_report(llm_md, rules_md) +def inject_section_bridges_into_markdown(md: str, bridges: dict[str, str]) -> str: + """ + 在「## 一、」…「## 九、」各章标题行之后插入 ``#### 衔接分析(大模型)`` 段落。 + 自第九章向前替换,避免多次插入导致偏移错位。 + """ + out = md + for key in "九八七六五四三二一": + content = (bridges.get(key) or "").strip() + if not content: + continue + pat = re.compile(rf"^(## {key}、[^\n]*)\n", re.MULTILINE) + + def _repl(m: re.Match[str], _c: str = content) -> str: + return m.group(0) + "\n#### 衔接分析(大模型)\n\n" + _c + "\n\n" + + out = pat.sub(_repl, out, count=1) + return out + + def _flat_comment_texts(comment_rows: list[dict[str, str]]) -> list[str]: """全部非空评价正文(与报告统计同源)。""" out: list[str] = [] @@ -148,6 +167,7 @@ def get_default_report_config() -> dict[str, Any]: jcr, _ = _jd_crawler_modules() return { "llm_comment_sentiment": False, + "llm_section_bridges": False, "comment_focus_words": list(jcr.COMMENT_FOCUS_WORDS), "comment_scenario_groups": [ {"label": lbl, "triggers": list(trs)} @@ -326,6 +346,49 @@ def write_competitor_analysis_for_run_dir( report_config=eff_rc, llm_sentiment_section_md=llm_sentiment_md or None, ) + + bridge_record: dict[str, Any] = { + "schema_version": 1, + "attempted": False, + } + skip_bridge = os.environ.get( + "MA_SKIP_LLM_SECTION_BRIDGES", "" + ).strip().lower() in ("1", "true", "yes") + env_bridge = os.environ.get( + "MA_ENABLE_LLM_SECTION_BRIDGES", "" + ).strip().lower() in ("1", "true", "yes") + want_bridge = bool(eff_rc.get("llm_section_bridges")) or env_bridge + if want_bridge and not skip_bridge: + from .llm_generate import ( + generate_section_bridges_llm, + split_competitor_report_for_bridges, + ) + + parts = split_competitor_report_for_bridges(md) + if parts: + bridge_record["attempted"] = True + try: + bridges = generate_section_bridges_llm( + keyword=kw, brief=brief_final, sections=parts + ) + bridge_record["keys_received"] = sorted(bridges.keys()) + md = inject_section_bridges_into_markdown(md, bridges) + bridge_record["ok"] = True + except Exception as e: + bridge_record["ok"] = False + bridge_record["error"] = str(e) + else: + bridge_record["skipped"] = "no_h2_sections_matched" + elif skip_bridge: + bridge_record["skipped"] = "MA_SKIP_LLM_SECTION_BRIDGES" + elif not want_bridge: + bridge_record["skipped"] = "not_enabled" + + (run_dir / "section_bridge_llm.json").write_text( + json.dumps(bridge_record, ensure_ascii=False, indent=2), + encoding="utf-8", + ) + out_md = run_dir / "competitor_analysis.md" out_md.write_text(md, encoding="utf-8") return run_dir diff --git a/backend/pipeline/llm_generate.py b/backend/pipeline/llm_generate.py index 41fae4b..d460b1e 100644 --- a/backend/pipeline/llm_generate.py +++ b/backend/pipeline/llm_generate.py @@ -5,6 +5,7 @@ from __future__ import annotations import json +import re import sys from pathlib import Path from typing import Any @@ -110,6 +111,113 @@ def generate_comment_sentiment_analysis_llm(payload: dict[str, Any]) -> str: return _call_llm(SENTIMENT_LLM_SYSTEM, user) +def split_competitor_report_for_bridges( + md: str, *, max_excerpt: int = 1200 +) -> dict[str, dict[str, str]]: + """ + 按「## 一、」…「## 九、」切分规则报告,供大模型按章写衔接分析。 + 每键含完整标题行与正文摘录(过长截断)。 + """ + pat = re.compile(r"^## ([一二三四五六七八九])、([^\n]*)$", re.MULTILINE) + matches = list(pat.finditer(md)) + out: dict[str, dict[str, str]] = {} + for i, m in enumerate(matches): + key = m.group(1) + rest = m.group(2) + title = f"## {key}、{rest}" + start = m.end() + end = matches[i + 1].start() if i + 1 < len(matches) else len(md) + body = md[start:end].strip() + exc = body[:max_excerpt] + if len(body) > max_excerpt: + exc += "\n\n…(本节摘录已截断)\n" + out[key] = {"title": title, "excerpt": exc} + return out + + +def _parse_llm_json_object(text: str) -> dict[str, Any]: + raw = (text or "").strip() + if not raw: + return {} + if raw.startswith("```"): + raw = re.sub(r"^```(?:json)?\s*", "", raw, flags=re.IGNORECASE) + raw = re.sub(r"\s*```\s*$", "", raw) + try: + obj = json.loads(raw) + return obj if isinstance(obj, dict) else {} + except json.JSONDecodeError: + pass + m = re.search(r"\{[\s\S]*\}", raw) + if m: + try: + obj = json.loads(m.group(0)) + return obj if isinstance(obj, dict) else {} + except json.JSONDecodeError: + pass + return {} + + +def _normalize_section_bridge_map(d: dict[str, Any]) -> dict[str, str]: + allowed = frozenset("一二三四五六七八九") + out: dict[str, str] = {} + for k, v in d.items(): + if not isinstance(k, str) or len(k) != 1 or k not in allowed: + continue + if isinstance(v, str) and v.strip(): + out[k] = v.strip() + return out + + +BRIDGE_SECTIONS_SYSTEM = """你是竞品监测报告的**章节衔接**撰稿助手。 + +**输入 JSON** 含: +- ``keyword``:监测词; +- ``competitor_brief``:与本报告一致的**结构化摘要**(已裁剪体积); +- ``sections``:键为汉字「一」~「九」,每项含 ``title``(该章完整二级标题行)与 ``excerpt``(该章正文开头摘录,可能已截断)。 + +**任务**:为 **sections 中出现的每一键** 各写一段 **衔接性分析**(帮读者从摘要与摘录过渡到读该章表格/图),并与 ``competitor_brief`` 中的数字与结论一致。 + +**硬性要求**: +- **仅输出一个 UTF-8 JSON 对象**(不要用 markdown 代码围栏包裹整段输出); +- 键必须为「一」「二」…「九」之一,且 **只对输入 sections 里存在的键** 给出字符串值;可省略无材料的键; +- 每个值为 **Markdown 片段**(约 3~10 句中文),**禁止**使用 ``## `` 开头的行(不要写新的二级章标题);可使用 ``###`` / ``####`` 或加粗小标题; +- 所有**定量表述**须能在 ``competitor_brief`` 或对应 ``excerpt`` 中找到依据,**禁止编造** SKU 数、份额、价格; +- 不要复述整章表格;不要写「详见下文矩阵」以外的空洞套话;可点出该章阅读重点(如价盘带、矩阵细类、评价规则局限等)。""" + + +def generate_section_bridges_llm( + *, + keyword: str, + brief: dict[str, Any], + sections: dict[str, dict[str, str]], +) -> dict[str, str]: + """一次 LLM 调用,返回各章衔接 Markdown 片段(键:一~九)。""" + if not sections: + return {} + compact = compact_brief_for_llm(brief, max_chars=100_000) + sec: dict[str, dict[str, str]] = { + k: {"title": v.get("title", ""), "excerpt": v.get("excerpt", "")} + for k, v in sections.items() + if isinstance(v, dict) + } + for max_exc in (1200, 900, 600, 400, 280): + for v in sec.values(): + ex = v.get("excerpt") or "" + if len(ex) > max_exc: + v["excerpt"] = ex[:max_exc] + "\n…\n" + payload = { + "keyword": keyword, + "competitor_brief": compact, + "sections": sec, + } + raw = json.dumps(payload, ensure_ascii=False) + if len(raw) <= 92_000: + break + user = "请严格按系统说明,**只输出一个 JSON 对象**(键为一~九,值为 Markdown 字符串):\n\n" + raw + text = _call_llm(BRIDGE_SECTIONS_SYSTEM, user) + return _normalize_section_bridge_map(_parse_llm_json_object(text)) + + STRATEGY_SYSTEM = """你是市场策略顾问,根据**结构化监测摘要**与业务侧填写的**决策字段**,把「规则底稿」润色为可读的策略 Markdown。 **规则**: diff --git a/backend/pipeline/serializers.py b/backend/pipeline/serializers.py index d7aa05f..050af0d 100644 --- a/backend/pipeline/serializers.py +++ b/backend/pipeline/serializers.py @@ -11,6 +11,7 @@ from .models import JdProduct, JdProductSnapshot, JobStatus, PipelineJob _REPORT_CONFIG_ALLOWED_KEYS = frozenset( { "llm_comment_sentiment", + "llm_section_bridges", "comment_focus_words", "comment_scenario_groups", "external_market_table_rows", @@ -29,6 +30,9 @@ def validate_report_config_body(value: dict) -> dict: if "llm_comment_sentiment" in value and value["llm_comment_sentiment"] is not None: if not isinstance(value["llm_comment_sentiment"], bool): raise serializers.ValidationError("llm_comment_sentiment 须为 true 或 false") + if "llm_section_bridges" in value and value["llm_section_bridges"] is not None: + if not isinstance(value["llm_section_bridges"], bool): + raise serializers.ValidationError("llm_section_bridges 须为 true 或 false") raw = json.dumps(value, ensure_ascii=False) if len(raw) > 120_000: raise serializers.ValidationError("报告配置体积过大") @@ -41,6 +45,7 @@ _ARTIFACT_FILES: tuple[tuple[str, str], ...] = ( ("comments", "comments_flat.csv"), ("detail_ware", "detail_ware_export.csv"), ("report", "competitor_analysis.md"), + ("section_bridge_llm", "section_bridge_llm.json"), ) diff --git a/frontend/src/composables/useReportConfigForm.js b/frontend/src/composables/useReportConfigForm.js index 1261865..92dce02 100644 --- a/frontend/src/composables/useReportConfigForm.js +++ b/frontend/src/composables/useReportConfigForm.js @@ -20,11 +20,15 @@ export function useReportConfigForm() { const marketRows = ref([ { indicator: '', value_and_scope: '', source: '', year: '' }, ]) + const useLlmCommentSentiment = ref(false) + const useLlmSectionBridges = ref(false) function resetToEmpty() { focusWordRows.value = [{ text: '' }] scenarioGroups.value = [{ label: '', triggersText: '' }] marketRows.value = [{ indicator: '', value_and_scope: '', source: '', year: '' }] + useLlmCommentSentiment.value = false + useLlmSectionBridges.value = false } /** @@ -94,6 +98,9 @@ export function useReportConfigForm() { } else { marketRows.value = [{ indicator: '', value_and_scope: '', source: '', year: '' }] } + + useLlmCommentSentiment.value = Boolean(cfg.llm_comment_sentiment) + useLlmSectionBridges.value = Boolean(cfg.llm_section_bridges) } /** @returns {Record} 可 PATCH 到后端的 report_config;全空则为 {} */ @@ -133,6 +140,9 @@ export function useReportConfigForm() { })) } + out.llm_comment_sentiment = useLlmCommentSentiment.value + out.llm_section_bridges = useLlmSectionBridges.value + return out } @@ -178,6 +188,8 @@ export function useReportConfigForm() { focusWordRows, scenarioGroups, marketRows, + useLlmCommentSentiment, + useLlmSectionBridges, resetToEmpty, applyFromApiConfig, buildPayload, diff --git a/frontend/src/views/jd/JdAnalysisBuildView.vue b/frontend/src/views/jd/JdAnalysisBuildView.vue index b853958..4f7f770 100644 --- a/frontend/src/views/jd/JdAnalysisBuildView.vue +++ b/frontend/src/views/jd/JdAnalysisBuildView.vue @@ -31,6 +31,8 @@ const { focusWordRows, scenarioGroups, marketRows, + useLlmCommentSentiment, + useLlmSectionBridges, applyFromApiConfig, buildPayload, addFocusRow, @@ -247,8 +249,18 @@ watch(

报告里的评价统计怎么算

- 下面三项都可以不改:留空并保存,表示沿用系统内置规则。请先点「保存以上设置」,再点「重新生成报告」(需要大模型时先勾选页面上方对应选项)。 + 下面几项都可以不改:留空并保存,表示沿用系统内置规则。请先点「保存以上设置」,再点「重新生成报告」;需要大模型时勾选下方「评价归纳」或「各章衔接」,页顶「使用大模型生成」仍用于整份报告另一种生成模式。

+
+ + +