mirror of
https://github.com/primedigitaltech/market-assistant.git
synced 2026-07-24 17:21:40 +08:00
feat(jd-pipeline): matrix chart and brief use totalSales for sales volume
Made-with: Cursor
This commit is contained in:
parent
2473a7cdac
commit
8611e78a26
@ -1578,7 +1578,7 @@ def _competitor_matrix_md_line(
|
|||||||
)
|
)
|
||||||
cat = _md_cell(_detail_category_path_cell(row), 24)
|
cat = _md_cell(_detail_category_path_cell(row), 24)
|
||||||
ing = _matrix_ingredients_cell(row)
|
ing = _matrix_ingredients_cell(row)
|
||||||
cc = _md_cell(_cell(row, "评价量(commentFuzzy)"), 10)
|
cc = _md_cell(_cell(row, "销量口径(totalSales)", "评价量(commentFuzzy)"), 14)
|
||||||
prev = _md_cell(_cell(row, "comment_preview"), 72)
|
prev = _md_cell(_cell(row, "comment_preview"), 72)
|
||||||
return (
|
return (
|
||||||
f"| {sku} | {title} | {brand} | {pj} | {df} | {shop} | {sell} | {rank} | "
|
f"| {sku} | {title} | {brand} | {pj} | {df} | {shop} | {sell} | {rank} | "
|
||||||
@ -1664,10 +1664,10 @@ def _focus_scenario_combo_bar_filename(group: str, index: int) -> str:
|
|||||||
return f"chart_focus_and_scenarios_bar__{slug}.png"
|
return f"chart_focus_and_scenarios_bar__{slug}.png"
|
||||||
|
|
||||||
|
|
||||||
def _matrix_prices_reviews_chart_filename(group: str, index: int) -> str:
|
def _matrix_prices_sales_chart_filename(group: str, index: int) -> str:
|
||||||
"""与 ``report_charts.generate_report_charts`` 中 ``chart_matrix_prices_reviews__*`` 一致。"""
|
"""与 ``report_charts.generate_report_charts`` 中 ``chart_matrix_prices_sales__*`` 一致。"""
|
||||||
slug = _scenario_group_asset_slug(group, index)
|
slug = _scenario_group_asset_slug(group, index)
|
||||||
return f"chart_matrix_prices_reviews__{slug}.png"
|
return f"chart_matrix_prices_sales__{slug}.png"
|
||||||
|
|
||||||
|
|
||||||
def _lines_4_reading_brand(
|
def _lines_4_reading_brand(
|
||||||
@ -2219,8 +2219,8 @@ def build_competitor_markdown(
|
|||||||
"分组**仅**使用合并表列 ``detail_category_path``(商详类目路径):**三级路径**取中间一段(如 … > **饼干** > 粗粮饼干),"
|
"分组**仅**使用合并表列 ``detail_category_path``(商详类目路径):**三级路径**取中间一段(如 … > **饼干** > 粗粮饼干),"
|
||||||
"**四级及以上**取倒数第二段(如 … > **面条** > 挂面)。**该列为空**或路径段均为内部编码、**无法解析出可读细类**的 SKU **不进入**本矩阵,亦**不参与**第八章按细类的评价统计。",
|
"**四级及以上**取倒数第二段(如 … > **面条** > 挂面)。**该列为空**或路径段均为内部编码、**无法解析出可读细类**的 SKU **不进入**本矩阵,亦**不参与**第八章按细类的评价统计。",
|
||||||
"",
|
"",
|
||||||
"**读图方式**:每个细类下为**并列横向条形图**(左:**展示价**(元);右:**评价量**(搜索列表侧文案,作**声量 proxy**,**非**平台销量/动销)),"
|
"**读图方式**:每个细类下为**并列横向条形图**(左:**展示价**(元);右:**销量**(搜索列表 ``totalSales`` 文案解析件数,如「已售50万+」计为 **50 万**)),"
|
||||||
"纵轴为**产品标题**(与 ``report_assets/chart_matrix_prices_reviews__*.png`` 同源)。**SKU、店铺、配料与评价摘要等明细不列入正文**,见本批次 ``keyword_pipeline_merged.csv``。",
|
"纵轴为**产品标题**(与 ``report_assets/chart_matrix_prices_sales__*.png`` 同源)。**SKU、店铺、配料与评价摘要等明细不列入正文**,见本批次 ``keyword_pipeline_merged.csv``。",
|
||||||
"",
|
"",
|
||||||
]
|
]
|
||||||
)
|
)
|
||||||
@ -2237,12 +2237,12 @@ def build_competitor_markdown(
|
|||||||
for gi, (gname, grows) in enumerate(grouped_matrix):
|
for gi, (gname, grows) in enumerate(grouped_matrix):
|
||||||
lines.append(f"### {gname}(**{len(grows)}** 款)")
|
lines.append(f"### {gname}(**{len(grows)}** 款)")
|
||||||
lines.append("")
|
lines.append("")
|
||||||
mx_chart = _matrix_prices_reviews_chart_filename(gname, gi)
|
mx_chart = _matrix_prices_sales_chart_filename(gname, gi)
|
||||||
lines.extend(
|
lines.extend(
|
||||||
_embed_chart(
|
_embed_chart(
|
||||||
run_dir,
|
run_dir,
|
||||||
mx_chart,
|
mx_chart,
|
||||||
f"「{_md_cell(gname, 20)}」· 展示价与评价量(搜索侧声量 proxy);纵轴为产品标题。",
|
f"「{_md_cell(gname, 20)}」· 展示价与销量(totalSales 解析);纵轴为产品标题。",
|
||||||
)
|
)
|
||||||
)
|
)
|
||||||
if not (run_dir / "report_assets" / mx_chart).is_file():
|
if not (run_dir / "report_assets" / mx_chart).is_file():
|
||||||
@ -2665,6 +2665,7 @@ def build_competitor_brief(
|
|||||||
"category": _detail_category_path_cell(row),
|
"category": _detail_category_path_cell(row),
|
||||||
"selling_point": _cell(row, "卖点(sellingPoint)")[:240],
|
"selling_point": _cell(row, "卖点(sellingPoint)")[:240],
|
||||||
"comment_fuzzy": _cell(row, "评价量(commentFuzzy)"),
|
"comment_fuzzy": _cell(row, "评价量(commentFuzzy)"),
|
||||||
|
"total_sales": _cell(row, "销量口径(totalSales)"),
|
||||||
}
|
}
|
||||||
)
|
)
|
||||||
matrix_groups.append(
|
matrix_groups.append(
|
||||||
|
|||||||
@ -200,6 +200,7 @@ _MERGED_LEAN_FIELDNAMES: tuple[str, ...] = (
|
|||||||
"榜单类文案(标签/腰带/标题数组中的榜、TOP 等)",
|
"榜单类文案(标签/腰带/标题数组中的榜、TOP 等)",
|
||||||
"评价量(commentFuzzy)",
|
"评价量(commentFuzzy)",
|
||||||
"销量楼层(commentSalesFloor)",
|
"销量楼层(commentSalesFloor)",
|
||||||
|
"销量口径(totalSales)",
|
||||||
"店铺名(shopName)",
|
"店铺名(shopName)",
|
||||||
"商品链接(toUrl,clickUrl,item.m.jd.com)",
|
"商品链接(toUrl,clickUrl,item.m.jd.com)",
|
||||||
"主图(imageurl,imageUrl)",
|
"主图(imageurl,imageUrl)",
|
||||||
|
|||||||
@ -339,6 +339,7 @@ JD_ITEM_CSV_FIELDS = (
|
|||||||
"original_price",
|
"original_price",
|
||||||
"selling_point",
|
"selling_point",
|
||||||
"comment_sales_floor",
|
"comment_sales_floor",
|
||||||
|
"total_sales",
|
||||||
"hot_list_rank",
|
"hot_list_rank",
|
||||||
"comment_count",
|
"comment_count",
|
||||||
"shop_name",
|
"shop_name",
|
||||||
@ -369,6 +370,7 @@ JD_EXPORT_COLUMN_HEADERS: dict[str, str] = {
|
|||||||
"original_price": "原价(oriPrice,originalPrice,marketPrice)",
|
"original_price": "原价(oriPrice,originalPrice,marketPrice)",
|
||||||
"selling_point": "卖点(sellingPoint)",
|
"selling_point": "卖点(sellingPoint)",
|
||||||
"comment_sales_floor": "销量楼层(commentSalesFloor)",
|
"comment_sales_floor": "销量楼层(commentSalesFloor)",
|
||||||
|
"total_sales": "销量口径(totalSales)",
|
||||||
"hot_list_rank": "榜单类文案(标签/腰带/标题数组中的榜、TOP 等)",
|
"hot_list_rank": "榜单类文案(标签/腰带/标题数组中的榜、TOP 等)",
|
||||||
"comment_count": "评价量(commentFuzzy)",
|
"comment_count": "评价量(commentFuzzy)",
|
||||||
"shop_name": "店铺名(shopName)",
|
"shop_name": "店铺名(shopName)",
|
||||||
@ -1239,6 +1241,13 @@ def _normalize_jd_api_row(d: dict[str, Any], *, keyword: str, page: int) -> dict
|
|||||||
120,
|
120,
|
||||||
)
|
)
|
||||||
comment_sales_floor = _jd_format_comment_sales_floor(d)
|
comment_sales_floor = _jd_format_comment_sales_floor(d)
|
||||||
|
total_sales = _human_text(
|
||||||
|
_sval_jd(
|
||||||
|
d,
|
||||||
|
("totalSales", "total_sales", "TotalSales"),
|
||||||
|
),
|
||||||
|
400,
|
||||||
|
)
|
||||||
|
|
||||||
tag_strs = _jd_iter_tag_strings(d)
|
tag_strs = _jd_iter_tag_strings(d)
|
||||||
arr_ind, _arr_groups = _jd_collect_list_string_fragments(
|
arr_ind, _arr_groups = _jd_collect_list_string_fragments(
|
||||||
@ -1301,6 +1310,7 @@ def _normalize_jd_api_row(d: dict[str, Any], *, keyword: str, page: int) -> dict
|
|||||||
out["original_price"] = orig
|
out["original_price"] = orig
|
||||||
out["selling_point"] = selling_point
|
out["selling_point"] = selling_point
|
||||||
out["comment_sales_floor"] = comment_sales_floor
|
out["comment_sales_floor"] = comment_sales_floor
|
||||||
|
out["total_sales"] = total_sales
|
||||||
out["hot_list_rank"] = hot_list_rank
|
out["hot_list_rank"] = hot_list_rank
|
||||||
out["comment_count"] = comment_count
|
out["comment_count"] = comment_count
|
||||||
out["shop_name"] = shop
|
out["shop_name"] = shop
|
||||||
|
|||||||
@ -14,6 +14,7 @@ JD_SEARCH_INTERNAL_KEYS: tuple[str, ...] = (
|
|||||||
"original_price",
|
"original_price",
|
||||||
"selling_point",
|
"selling_point",
|
||||||
"comment_sales_floor",
|
"comment_sales_floor",
|
||||||
|
"total_sales",
|
||||||
"hot_list_rank",
|
"hot_list_rank",
|
||||||
"comment_count",
|
"comment_count",
|
||||||
"shop_name",
|
"shop_name",
|
||||||
@ -39,6 +40,7 @@ JD_SEARCH_CSV_HEADERS: dict[str, str] = {
|
|||||||
"original_price": "原价(oriPrice,originalPrice,marketPrice)",
|
"original_price": "原价(oriPrice,originalPrice,marketPrice)",
|
||||||
"selling_point": "卖点(sellingPoint)",
|
"selling_point": "卖点(sellingPoint)",
|
||||||
"comment_sales_floor": "销量楼层(commentSalesFloor)",
|
"comment_sales_floor": "销量楼层(commentSalesFloor)",
|
||||||
|
"total_sales": "销量口径(totalSales)",
|
||||||
"hot_list_rank": "榜单类文案(标签/腰带/标题数组中的榜、TOP 等)",
|
"hot_list_rank": "榜单类文案(标签/腰带/标题数组中的榜、TOP 等)",
|
||||||
"comment_count": "评价量(commentFuzzy)",
|
"comment_count": "评价量(commentFuzzy)",
|
||||||
"shop_name": "店铺名(shopName)",
|
"shop_name": "店铺名(shopName)",
|
||||||
@ -117,6 +119,7 @@ MERGED_SEARCH_CSV_COLUMNS: tuple[str, ...] = (
|
|||||||
"榜单类文案(标签/腰带/标题数组中的榜、TOP 等)",
|
"榜单类文案(标签/腰带/标题数组中的榜、TOP 等)",
|
||||||
"评价量(commentFuzzy)",
|
"评价量(commentFuzzy)",
|
||||||
"销量楼层(commentSalesFloor)",
|
"销量楼层(commentSalesFloor)",
|
||||||
|
"销量口径(totalSales)",
|
||||||
"店铺名(shopName)",
|
"店铺名(shopName)",
|
||||||
"商品链接(toUrl,clickUrl,item.m.jd.com)",
|
"商品链接(toUrl,clickUrl,item.m.jd.com)",
|
||||||
"主图(imageurl,imageUrl)",
|
"主图(imageurl,imageUrl)",
|
||||||
@ -138,6 +141,7 @@ MERGED_SEARCH_INTERNAL_KEYS: tuple[str, ...] = (
|
|||||||
"hot_list_rank",
|
"hot_list_rank",
|
||||||
"comment_fuzzy",
|
"comment_fuzzy",
|
||||||
"comment_sales_floor",
|
"comment_sales_floor",
|
||||||
|
"total_sales",
|
||||||
"shop_name",
|
"shop_name",
|
||||||
"detail_url",
|
"detail_url",
|
||||||
"image",
|
"image",
|
||||||
|
|||||||
@ -0,0 +1,18 @@
|
|||||||
|
# Generated manually for keyword_pipeline_merged.csv column 销量口径(totalSales).
|
||||||
|
|
||||||
|
from django.db import migrations, models
|
||||||
|
|
||||||
|
|
||||||
|
class Migration(migrations.Migration):
|
||||||
|
|
||||||
|
dependencies = [
|
||||||
|
("pipeline", "0013_rebuild_pipelinejobcheckpoint"),
|
||||||
|
]
|
||||||
|
|
||||||
|
operations = [
|
||||||
|
migrations.AddField(
|
||||||
|
model_name="jdjobmergedrow",
|
||||||
|
name="total_sales",
|
||||||
|
field=models.TextField(blank=True, default=""),
|
||||||
|
),
|
||||||
|
]
|
||||||
@ -277,6 +277,7 @@ class JdJobMergedRow(models.Model):
|
|||||||
hot_list_rank = models.TextField(blank=True, default="")
|
hot_list_rank = models.TextField(blank=True, default="")
|
||||||
comment_fuzzy = models.TextField(blank=True, default="")
|
comment_fuzzy = models.TextField(blank=True, default="")
|
||||||
comment_sales_floor = models.TextField(blank=True, default="")
|
comment_sales_floor = models.TextField(blank=True, default="")
|
||||||
|
total_sales = models.TextField(blank=True, default="")
|
||||||
shop_name = models.TextField(blank=True, default="")
|
shop_name = models.TextField(blank=True, default="")
|
||||||
detail_url = models.TextField(blank=True, default="")
|
detail_url = models.TextField(blank=True, default="")
|
||||||
image = models.TextField(blank=True, default="")
|
image = models.TextField(blank=True, default="")
|
||||||
|
|||||||
@ -80,11 +80,17 @@ def _float_price_from_cell(s: str) -> float | None:
|
|||||||
return None
|
return None
|
||||||
|
|
||||||
|
|
||||||
def _reviews_volume_int(s: str) -> int:
|
def _cn_volume_int(s: str) -> int:
|
||||||
"""与矩阵「评价量」列同源:从搜索侧模糊文案中抽取整数(含「万」)。"""
|
"""
|
||||||
|
从搜索侧文案抽取非负整数(评价量/销量等):支持「亿」「万」及纯数字;
|
||||||
|
如 ``已售50万+ | good:99%好评`` → 500000。
|
||||||
|
"""
|
||||||
t = (s or "").strip().replace(",", "").replace(",", "")
|
t = (s or "").strip().replace(",", "").replace(",", "")
|
||||||
if not t:
|
if not t:
|
||||||
return 0
|
return 0
|
||||||
|
m = re.search(r"(\d+(?:\.\d+)?)\s*亿", t)
|
||||||
|
if m:
|
||||||
|
return int(round(float(m.group(1)) * 100_000_000))
|
||||||
m = re.search(r"(\d+(?:\.\d+)?)\s*万", t)
|
m = re.search(r"(\d+(?:\.\d+)?)\s*万", t)
|
||||||
if m:
|
if m:
|
||||||
return int(round(float(m.group(1)) * 10_000))
|
return int(round(float(m.group(1)) * 10_000))
|
||||||
@ -97,7 +103,7 @@ def _reviews_volume_int(s: str) -> int:
|
|||||||
def _format_xaxis_int_cn(x: float, _pos: int | None) -> str:
|
def _format_xaxis_int_cn(x: float, _pos: int | None) -> str:
|
||||||
"""
|
"""
|
||||||
横轴大整数刻度:用「万」「亿」表述,避免 matplotlib 默认 ``1e6`` 科学计数法。
|
横轴大整数刻度:用「万」「亿」表述,避免 matplotlib 默认 ``1e6`` 科学计数法。
|
||||||
用于评价量、条数等非负计数。
|
用于销量、评价量、条数等非负计数。
|
||||||
"""
|
"""
|
||||||
if not math.isfinite(x):
|
if not math.isfinite(x):
|
||||||
return ""
|
return ""
|
||||||
@ -589,20 +595,20 @@ def generate_report_charts(run_dir: Path, brief: dict[str, Any]) -> list[str]:
|
|||||||
p = _float_price_from_cell(str(s.get(k) or ""))
|
p = _float_price_from_cell(str(s.get(k) or ""))
|
||||||
if p is not None:
|
if p is not None:
|
||||||
break
|
break
|
||||||
rev = _reviews_volume_int(str(s.get("comment_fuzzy") or ""))
|
sales = _cn_volume_int(str(s.get("total_sales") or ""))
|
||||||
rows_data.append((label, p, rev))
|
rows_data.append((label, p, sales))
|
||||||
rows_data.sort(key=lambda x: x[0])
|
rows_data.sort(key=lambda x: x[0])
|
||||||
if not rows_data:
|
if not rows_data:
|
||||||
continue
|
continue
|
||||||
if not any(
|
if not any(
|
||||||
(pr is not None and pr > 0) or rv > 0
|
(pr is not None and pr > 0) or sv > 0
|
||||||
for _, pr, rv in rows_data
|
for _, pr, sv in rows_data
|
||||||
):
|
):
|
||||||
continue
|
continue
|
||||||
n = len(rows_data)
|
n = len(rows_data)
|
||||||
labels_mx = [x[0] for x in rows_data]
|
labels_mx = [x[0] for x in rows_data]
|
||||||
prices_mx = [x[1] for x in rows_data]
|
prices_mx = [x[1] for x in rows_data]
|
||||||
reviews_mx = [x[2] for x in rows_data]
|
sales_mx = [x[2] for x in rows_data]
|
||||||
y_pos = list(range(n))
|
y_pos = list(range(n))
|
||||||
fig_h = max(3.4, min(14.0, 0.38 * n + 2.4))
|
fig_h = max(3.4, min(14.0, 0.38 * n + 2.4))
|
||||||
fig, (ax_l, ax_r) = plt.subplots(
|
fig, (ax_l, ax_r) = plt.subplots(
|
||||||
@ -616,21 +622,21 @@ def generate_report_charts(run_dir: Path, brief: dict[str, Any]) -> list[str]:
|
|||||||
ax_l.invert_yaxis()
|
ax_l.invert_yaxis()
|
||||||
ax_l.set_xlabel("展示价(元)", fontsize=9)
|
ax_l.set_xlabel("展示价(元)", fontsize=9)
|
||||||
ax_l.set_title("展示价", fontsize=10, pad=8)
|
ax_l.set_title("展示价", fontsize=10, pad=8)
|
||||||
ax_r.barh(y_pos, reviews_mx, height=0.62, color="#059669")
|
ax_r.barh(y_pos, sales_mx, height=0.62, color="#059669")
|
||||||
ax_r.set_xlabel("评价量(搜索侧,单位:条)", fontsize=9)
|
ax_r.set_xlabel("销量(搜索列表 totalSales 口径,已解析为件数)", fontsize=9)
|
||||||
ax_r.set_title("评价量 / 声量", fontsize=10, pad=8)
|
ax_r.set_title("销量", fontsize=10, pad=8)
|
||||||
ax_r.xaxis.set_major_formatter(
|
ax_r.xaxis.set_major_formatter(
|
||||||
FuncFormatter(_format_xaxis_int_cn)
|
FuncFormatter(_format_xaxis_int_cn)
|
||||||
)
|
)
|
||||||
ax_r.tick_params(axis="y", left=False, labelleft=False)
|
ax_r.tick_params(axis="y", left=False, labelleft=False)
|
||||||
ttl = gname[:22] if gname else "细类"
|
ttl = gname[:22] if gname else "细类"
|
||||||
fig.suptitle(
|
fig.suptitle(
|
||||||
f"「{ttl}」· 竞品矩阵:价格与评价量(与 §5 表同源)",
|
f"「{ttl}」· 竞品矩阵:价格与销量(与 §5 表同源)",
|
||||||
fontsize=11,
|
fontsize=11,
|
||||||
y=1.01,
|
y=1.01,
|
||||||
)
|
)
|
||||||
fig.tight_layout()
|
fig.tight_layout()
|
||||||
out_mx = out_dir / f"chart_matrix_prices_reviews__{slug}.png"
|
out_mx = out_dir / f"chart_matrix_prices_sales__{slug}.png"
|
||||||
fig.savefig(out_mx, dpi=130, bbox_inches="tight")
|
fig.savefig(out_mx, dpi=130, bbox_inches="tight")
|
||||||
plt.close(fig)
|
plt.close(fig)
|
||||||
created.append(out_mx.name)
|
created.append(out_mx.name)
|
||||||
|
|||||||
@ -8,7 +8,6 @@ from pathlib import Path
|
|||||||
from django.conf import settings
|
from django.conf import settings
|
||||||
from django.test import SimpleTestCase
|
from django.test import SimpleTestCase
|
||||||
|
|
||||||
from pipeline.csv_schema import infer_total_sales_from_sales_floor
|
|
||||||
from pipeline.report_charts import _cn_volume_int
|
from pipeline.report_charts import _cn_volume_int
|
||||||
|
|
||||||
|
|
||||||
@ -179,13 +178,6 @@ class BuildCompetitorBriefTests(SimpleTestCase):
|
|||||||
)
|
)
|
||||||
self.assertEqual(_cn_volume_int("2.5亿件"), 250_000_000)
|
self.assertEqual(_cn_volume_int("2.5亿件"), 250_000_000)
|
||||||
|
|
||||||
def test_infer_total_sales_from_sales_floor(self) -> None:
|
|
||||||
self.assertEqual(
|
|
||||||
infer_total_sales_from_sales_floor("good:99%好评 | 已售50万+"),
|
|
||||||
"已售50万+",
|
|
||||||
)
|
|
||||||
self.assertEqual(infer_total_sales_from_sales_floor(""), "")
|
|
||||||
|
|
||||||
def test_mix_top_remainder_sums_to_all_rows(self) -> None:
|
def test_mix_top_remainder_sums_to_all_rows(self) -> None:
|
||||||
"""饼图与 §4 表同源:mix_top 各 count 之和须等于含名行数。"""
|
"""饼图与 §4 表同源:mix_top 各 count 之和须等于含名行数。"""
|
||||||
root = Path(settings.CRAWLER_JD_ROOT).resolve()
|
root = Path(settings.CRAWLER_JD_ROOT).resolve()
|
||||||
|
|||||||
@ -82,7 +82,8 @@
|
|||||||
"shop": "示例店",
|
"shop": "示例店",
|
||||||
"category": "休闲食品 > 饼干 > 粗粮饼干",
|
"category": "休闲食品 > 饼干 > 粗粮饼干",
|
||||||
"selling_point": "",
|
"selling_point": "",
|
||||||
"comment_fuzzy": "20万+"
|
"comment_fuzzy": "1万+",
|
||||||
|
"total_sales": "已售50万+ | good:99%好评"
|
||||||
}
|
}
|
||||||
]
|
]
|
||||||
}
|
}
|
||||||
|
|||||||
Loading…
x
Reference in New Issue
Block a user