mirror of
https://github.com/primedigitaltech/market-assistant.git
synced 2026-07-25 17:51:37 +08:00
feat(dataset): 搜索/宽表支持按解析后销量与评价量排序
新增 sales_sort_value、comment_count_sort_value 与 volume_parse;入库与迁移回填;sort=total_sales|comment_count;前端排序下拉文案。 Made-with: Cursor
This commit is contained in:
parent
f21bfbe02b
commit
a43e1f9b1d
@ -8,7 +8,16 @@ from django.db.models.expressions import OrderBy
|
|||||||
from rest_framework.request import Request
|
from rest_framework.request import Request
|
||||||
|
|
||||||
SEARCH_SORT_FIELDS = frozenset(
|
SEARCH_SORT_FIELDS = frozenset(
|
||||||
{"row_index", "price", "sku_id", "title", "leaf_category", "matrix_group_label"}
|
{
|
||||||
|
"row_index",
|
||||||
|
"price",
|
||||||
|
"sku_id",
|
||||||
|
"title",
|
||||||
|
"leaf_category",
|
||||||
|
"matrix_group_label",
|
||||||
|
"total_sales",
|
||||||
|
"comment_count",
|
||||||
|
}
|
||||||
)
|
)
|
||||||
DETAIL_SORT_FIELDS = frozenset(
|
DETAIL_SORT_FIELDS = frozenset(
|
||||||
{
|
{
|
||||||
@ -29,6 +38,8 @@ MERGED_SORT_FIELDS = frozenset(
|
|||||||
"leaf_category",
|
"leaf_category",
|
||||||
"detail_category_path",
|
"detail_category_path",
|
||||||
"matrix_group_label",
|
"matrix_group_label",
|
||||||
|
"total_sales",
|
||||||
|
"comment_count",
|
||||||
}
|
}
|
||||||
)
|
)
|
||||||
|
|
||||||
@ -113,6 +124,16 @@ def apply_search_order(qs: QuerySet, sort: str, desc: bool) -> QuerySet:
|
|||||||
OrderBy(F("price_value"), descending=desc, nulls_last=True),
|
OrderBy(F("price_value"), descending=desc, nulls_last=True),
|
||||||
"row_index",
|
"row_index",
|
||||||
)
|
)
|
||||||
|
if sort == "total_sales":
|
||||||
|
return qs.order_by(
|
||||||
|
OrderBy(F("sales_sort_value"), descending=desc, nulls_last=True),
|
||||||
|
"row_index",
|
||||||
|
)
|
||||||
|
if sort == "comment_count":
|
||||||
|
return qs.order_by(
|
||||||
|
OrderBy(F("comment_count_sort_value"), descending=desc, nulls_last=True),
|
||||||
|
"row_index",
|
||||||
|
)
|
||||||
if sort == "row_index":
|
if sort == "row_index":
|
||||||
return qs.order_by(OrderBy(F("row_index"), descending=desc))
|
return qs.order_by(OrderBy(F("row_index"), descending=desc))
|
||||||
field = {
|
field = {
|
||||||
@ -191,6 +212,16 @@ def apply_merged_order(qs: QuerySet, sort: str, desc: bool) -> QuerySet:
|
|||||||
OrderBy(F("price_value"), descending=desc, nulls_last=True),
|
OrderBy(F("price_value"), descending=desc, nulls_last=True),
|
||||||
"row_index",
|
"row_index",
|
||||||
)
|
)
|
||||||
|
if sort == "total_sales":
|
||||||
|
return qs.order_by(
|
||||||
|
OrderBy(F("sales_sort_value"), descending=desc, nulls_last=True),
|
||||||
|
"row_index",
|
||||||
|
)
|
||||||
|
if sort == "comment_count":
|
||||||
|
return qs.order_by(
|
||||||
|
OrderBy(F("comment_count_sort_value"), descending=desc, nulls_last=True),
|
||||||
|
"row_index",
|
||||||
|
)
|
||||||
if sort == "row_index":
|
if sort == "row_index":
|
||||||
return qs.order_by(OrderBy(F("row_index"), descending=desc))
|
return qs.order_by(OrderBy(F("row_index"), descending=desc))
|
||||||
field = {
|
field = {
|
||||||
|
|||||||
@ -40,6 +40,11 @@ from .models import (
|
|||||||
PipelineJob,
|
PipelineJob,
|
||||||
)
|
)
|
||||||
from .price_parse import effective_list_price_value, float_price_from_cell
|
from .price_parse import effective_list_price_value, float_price_from_cell
|
||||||
|
from .volume_parse import (
|
||||||
|
comment_count_sort_value_from_cell,
|
||||||
|
comment_count_sort_value_from_merged,
|
||||||
|
sales_sort_value_from_search_cells,
|
||||||
|
)
|
||||||
|
|
||||||
logger = logging.getLogger(__name__)
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
@ -210,12 +215,18 @@ def ingest_job_dataset_rows(job: PipelineJob) -> dict[str, Any]:
|
|||||||
pv = effective_list_price_value(
|
pv = effective_list_price_value(
|
||||||
kw.get("coupon_price"), kw.get("price"), kw.get("original_price")
|
kw.get("coupon_price"), kw.get("price"), kw.get("original_price")
|
||||||
)
|
)
|
||||||
|
sv = sales_sort_value_from_search_cells(
|
||||||
|
kw.get("total_sales"), kw.get("comment_sales_floor")
|
||||||
|
)
|
||||||
|
cv = comment_count_sort_value_from_cell(kw.get("comment_count"))
|
||||||
s_objs.append(
|
s_objs.append(
|
||||||
JdJobSearchRow(
|
JdJobSearchRow(
|
||||||
job=job,
|
job=job,
|
||||||
row_index=i,
|
row_index=i,
|
||||||
matrix_group_label="",
|
matrix_group_label="",
|
||||||
price_value=pv,
|
price_value=pv,
|
||||||
|
sales_sort_value=sv,
|
||||||
|
comment_count_sort_value=cv,
|
||||||
**kw,
|
**kw,
|
||||||
)
|
)
|
||||||
)
|
)
|
||||||
@ -263,12 +274,20 @@ def ingest_job_dataset_rows(job: PipelineJob) -> dict[str, Any]:
|
|||||||
pv = effective_list_price_value(
|
pv = effective_list_price_value(
|
||||||
kw.get("coupon_price"), kw.get("price"), kw.get("original_price")
|
kw.get("coupon_price"), kw.get("price"), kw.get("original_price")
|
||||||
)
|
)
|
||||||
|
msv = sales_sort_value_from_search_cells(
|
||||||
|
kw.get("total_sales"), kw.get("comment_sales_floor")
|
||||||
|
)
|
||||||
|
mcv = comment_count_sort_value_from_merged(
|
||||||
|
kw.get("pipeline_comment_count")
|
||||||
|
)
|
||||||
m_objs.append(
|
m_objs.append(
|
||||||
JdJobMergedRow(
|
JdJobMergedRow(
|
||||||
job=job,
|
job=job,
|
||||||
row_index=i,
|
row_index=i,
|
||||||
matrix_group_label=mg,
|
matrix_group_label=mg,
|
||||||
price_value=pv,
|
price_value=pv,
|
||||||
|
sales_sort_value=msv,
|
||||||
|
comment_count_sort_value=mcv,
|
||||||
**kw,
|
**kw,
|
||||||
)
|
)
|
||||||
)
|
)
|
||||||
|
|||||||
@ -0,0 +1,99 @@
|
|||||||
|
# Generated by Django 5.2.1 on 2026-04-16 02:13
|
||||||
|
|
||||||
|
from django.db import migrations, models
|
||||||
|
|
||||||
|
|
||||||
|
def backfill_volume_sort_fields(apps, schema_editor):
|
||||||
|
from pipeline.volume_parse import (
|
||||||
|
comment_count_sort_value_from_cell,
|
||||||
|
comment_count_sort_value_from_merged,
|
||||||
|
sales_sort_value_from_search_cells,
|
||||||
|
)
|
||||||
|
|
||||||
|
JdJobSearchRow = apps.get_model("pipeline", "JdJobSearchRow")
|
||||||
|
JdJobMergedRow = apps.get_model("pipeline", "JdJobMergedRow")
|
||||||
|
|
||||||
|
chunk: list = []
|
||||||
|
for r in JdJobSearchRow.objects.all().iterator(chunk_size=400):
|
||||||
|
r.sales_sort_value = sales_sort_value_from_search_cells(
|
||||||
|
r.total_sales or "", r.comment_sales_floor or ""
|
||||||
|
)
|
||||||
|
r.comment_count_sort_value = comment_count_sort_value_from_cell(
|
||||||
|
r.comment_count or ""
|
||||||
|
)
|
||||||
|
chunk.append(r)
|
||||||
|
if len(chunk) >= 400:
|
||||||
|
JdJobSearchRow.objects.bulk_update(
|
||||||
|
chunk, ["sales_sort_value", "comment_count_sort_value"]
|
||||||
|
)
|
||||||
|
chunk.clear()
|
||||||
|
if chunk:
|
||||||
|
JdJobSearchRow.objects.bulk_update(
|
||||||
|
chunk, ["sales_sort_value", "comment_count_sort_value"]
|
||||||
|
)
|
||||||
|
|
||||||
|
chunk = []
|
||||||
|
for r in JdJobMergedRow.objects.all().iterator(chunk_size=400):
|
||||||
|
r.sales_sort_value = sales_sort_value_from_search_cells(
|
||||||
|
r.total_sales or "", r.comment_sales_floor or ""
|
||||||
|
)
|
||||||
|
r.comment_count_sort_value = comment_count_sort_value_from_merged(
|
||||||
|
r.pipeline_comment_count or ""
|
||||||
|
)
|
||||||
|
chunk.append(r)
|
||||||
|
if len(chunk) >= 400:
|
||||||
|
JdJobMergedRow.objects.bulk_update(
|
||||||
|
chunk, ["sales_sort_value", "comment_count_sort_value"]
|
||||||
|
)
|
||||||
|
chunk.clear()
|
||||||
|
if chunk:
|
||||||
|
JdJobMergedRow.objects.bulk_update(
|
||||||
|
chunk, ["sales_sort_value", "comment_count_sort_value"]
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class Migration(migrations.Migration):
|
||||||
|
|
||||||
|
dependencies = [
|
||||||
|
('pipeline', '0019_matrix_label_verbose_category'),
|
||||||
|
]
|
||||||
|
|
||||||
|
operations = [
|
||||||
|
migrations.AddField(
|
||||||
|
model_name='jdjobmergedrow',
|
||||||
|
name='comment_count_sort_value',
|
||||||
|
field=models.BigIntegerField(blank=True, db_index=True, help_text='从 pipeline_comment_count 解析,供排序', null=True),
|
||||||
|
),
|
||||||
|
migrations.AddField(
|
||||||
|
model_name='jdjobmergedrow',
|
||||||
|
name='sales_sort_value',
|
||||||
|
field=models.BigIntegerField(blank=True, db_index=True, help_text='从 total_sales / 销量楼层解析,供排序', null=True),
|
||||||
|
),
|
||||||
|
migrations.AddField(
|
||||||
|
model_name='jdjobsearchrow',
|
||||||
|
name='comment_count_sort_value',
|
||||||
|
field=models.BigIntegerField(blank=True, db_index=True, help_text='从评价量文案解析,供排序', null=True),
|
||||||
|
),
|
||||||
|
migrations.AddField(
|
||||||
|
model_name='jdjobsearchrow',
|
||||||
|
name='sales_sort_value',
|
||||||
|
field=models.BigIntegerField(blank=True, db_index=True, help_text='从 total_sales / 销量楼层解析,供排序', null=True),
|
||||||
|
),
|
||||||
|
migrations.AddIndex(
|
||||||
|
model_name='jdjobmergedrow',
|
||||||
|
index=models.Index(fields=['job', 'sales_sort_value'], name='pipeline_jd_job_id_b59c42_idx'),
|
||||||
|
),
|
||||||
|
migrations.AddIndex(
|
||||||
|
model_name='jdjobmergedrow',
|
||||||
|
index=models.Index(fields=['job', 'comment_count_sort_value'], name='pipeline_jd_job_id_2c9d52_idx'),
|
||||||
|
),
|
||||||
|
migrations.AddIndex(
|
||||||
|
model_name='jdjobsearchrow',
|
||||||
|
index=models.Index(fields=['job', 'sales_sort_value'], name='pipeline_jd_job_id_6798d3_idx'),
|
||||||
|
),
|
||||||
|
migrations.AddIndex(
|
||||||
|
model_name='jdjobsearchrow',
|
||||||
|
index=models.Index(fields=['job', 'comment_count_sort_value'], name='pipeline_jd_job_id_f0aace_idx'),
|
||||||
|
),
|
||||||
|
migrations.RunPython(backfill_volume_sort_fields, migrations.RunPython.noop),
|
||||||
|
]
|
||||||
@ -199,6 +199,18 @@ class JdJobSearchRow(models.Model):
|
|||||||
help_text="与 §5 矩阵同源:由合并表商详路径解析;可按 SKU 从合并表回填",
|
help_text="与 §5 矩阵同源:由合并表商详路径解析;可按 SKU 从合并表回填",
|
||||||
)
|
)
|
||||||
price_value = models.FloatField(null=True, blank=True, db_index=True)
|
price_value = models.FloatField(null=True, blank=True, db_index=True)
|
||||||
|
sales_sort_value = models.BigIntegerField(
|
||||||
|
null=True,
|
||||||
|
blank=True,
|
||||||
|
db_index=True,
|
||||||
|
help_text="从 total_sales / 销量楼层解析,供排序",
|
||||||
|
)
|
||||||
|
comment_count_sort_value = models.BigIntegerField(
|
||||||
|
null=True,
|
||||||
|
blank=True,
|
||||||
|
db_index=True,
|
||||||
|
help_text="从评价量文案解析,供排序",
|
||||||
|
)
|
||||||
platform = models.TextField(blank=True, default="")
|
platform = models.TextField(blank=True, default="")
|
||||||
keyword = models.TextField(blank=True, default="")
|
keyword = models.TextField(blank=True, default="")
|
||||||
page = models.TextField(blank=True, default="")
|
page = models.TextField(blank=True, default="")
|
||||||
@ -216,6 +228,8 @@ class JdJobSearchRow(models.Model):
|
|||||||
models.Index(fields=["job", "leaf_category_norm"]),
|
models.Index(fields=["job", "leaf_category_norm"]),
|
||||||
models.Index(fields=["job", "matrix_group_label"]),
|
models.Index(fields=["job", "matrix_group_label"]),
|
||||||
models.Index(fields=["job", "price_value"]),
|
models.Index(fields=["job", "price_value"]),
|
||||||
|
models.Index(fields=["job", "sales_sort_value"]),
|
||||||
|
models.Index(fields=["job", "comment_count_sort_value"]),
|
||||||
]
|
]
|
||||||
|
|
||||||
def __str__(self) -> str:
|
def __str__(self) -> str:
|
||||||
@ -344,6 +358,18 @@ class JdJobMergedRow(models.Model):
|
|||||||
help_text="与 §5 矩阵同源:由 detail_category_path 解析",
|
help_text="与 §5 矩阵同源:由 detail_category_path 解析",
|
||||||
)
|
)
|
||||||
price_value = models.FloatField(null=True, blank=True, db_index=True)
|
price_value = models.FloatField(null=True, blank=True, db_index=True)
|
||||||
|
sales_sort_value = models.BigIntegerField(
|
||||||
|
null=True,
|
||||||
|
blank=True,
|
||||||
|
db_index=True,
|
||||||
|
help_text="从 total_sales / 销量楼层解析,供排序",
|
||||||
|
)
|
||||||
|
comment_count_sort_value = models.BigIntegerField(
|
||||||
|
null=True,
|
||||||
|
blank=True,
|
||||||
|
db_index=True,
|
||||||
|
help_text="从 pipeline_comment_count 解析,供排序",
|
||||||
|
)
|
||||||
keyword = models.TextField(blank=True, default="")
|
keyword = models.TextField(blank=True, default="")
|
||||||
page = models.TextField(blank=True, default="")
|
page = models.TextField(blank=True, default="")
|
||||||
detail_brand = models.TextField(blank=True, default="")
|
detail_brand = models.TextField(blank=True, default="")
|
||||||
@ -370,6 +396,8 @@ class JdJobMergedRow(models.Model):
|
|||||||
models.Index(fields=["job", "leaf_category_norm"]),
|
models.Index(fields=["job", "leaf_category_norm"]),
|
||||||
models.Index(fields=["job", "matrix_group_label"]),
|
models.Index(fields=["job", "matrix_group_label"]),
|
||||||
models.Index(fields=["job", "price_value"]),
|
models.Index(fields=["job", "price_value"]),
|
||||||
|
models.Index(fields=["job", "sales_sort_value"]),
|
||||||
|
models.Index(fields=["job", "comment_count_sort_value"]),
|
||||||
]
|
]
|
||||||
|
|
||||||
def __str__(self) -> str:
|
def __str__(self) -> str:
|
||||||
|
|||||||
30
backend/pipeline/tests/test_volume_parse.py
Normal file
30
backend/pipeline/tests/test_volume_parse.py
Normal file
@ -0,0 +1,30 @@
|
|||||||
|
"""销量/评价量文案解析(与 reporting.charts._cn_volume_int 口径一致)。"""
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from django.test import SimpleTestCase
|
||||||
|
|
||||||
|
from pipeline.volume_parse import (
|
||||||
|
cn_volume_int,
|
||||||
|
comment_count_sort_value_from_cell,
|
||||||
|
sales_sort_value_from_search_cells,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class VolumeParseTests(SimpleTestCase):
|
||||||
|
def test_cn_volume_wan(self) -> None:
|
||||||
|
self.assertEqual(cn_volume_int("已售50万+"), 500_000)
|
||||||
|
|
||||||
|
def test_cn_volume_yi(self) -> None:
|
||||||
|
self.assertEqual(cn_volume_int("1.2亿件"), 120_000_000)
|
||||||
|
|
||||||
|
def test_sales_sort_prefers_total_sales(self) -> None:
|
||||||
|
v = sales_sort_value_from_search_cells("已售1万+", "已售5000+")
|
||||||
|
self.assertEqual(v, 10_000)
|
||||||
|
|
||||||
|
def test_sales_sort_fallback_floor(self) -> None:
|
||||||
|
v = sales_sort_value_from_search_cells("", "已售3万+")
|
||||||
|
self.assertEqual(v, 30_000)
|
||||||
|
|
||||||
|
def test_comment_count_cell(self) -> None:
|
||||||
|
v = comment_count_sort_value_from_cell("5000+条评价")
|
||||||
|
self.assertEqual(v, 5000)
|
||||||
50
backend/pipeline/volume_parse.py
Normal file
50
backend/pipeline/volume_parse.py
Normal file
@ -0,0 +1,50 @@
|
|||||||
|
"""从爬虫导出文案解析销量、评价量等非负整数(与 reporting.charts._cn_volume_int 同源)。"""
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import re
|
||||||
|
|
||||||
|
|
||||||
|
def cn_volume_int(s: str | None) -> int:
|
||||||
|
"""
|
||||||
|
支持「亿」「万」及纯数字;如 ``已售50万+`` → 500000。
|
||||||
|
无法解析时返回 0。
|
||||||
|
"""
|
||||||
|
t = (s or "").strip().replace(",", "").replace(",", "")
|
||||||
|
if not t:
|
||||||
|
return 0
|
||||||
|
m = re.search(r"(\d+(?:\.\d+)?)\s*亿", t)
|
||||||
|
if m:
|
||||||
|
return int(round(float(m.group(1)) * 100_000_000))
|
||||||
|
m = re.search(r"(\d+(?:\.\d+)?)\s*万", t)
|
||||||
|
if m:
|
||||||
|
return int(round(float(m.group(1)) * 10_000))
|
||||||
|
m2 = re.search(r"(\d+)", t)
|
||||||
|
if m2:
|
||||||
|
return int(m2.group(1))
|
||||||
|
return 0
|
||||||
|
|
||||||
|
|
||||||
|
def sales_sort_value_from_search_cells(total_sales: str, comment_sales_floor: str) -> int | None:
|
||||||
|
"""搜索行:优先 ``total_sales``,否则销量楼层;两列皆空则 None。"""
|
||||||
|
ts = (total_sales or "").strip()
|
||||||
|
fl = (comment_sales_floor or "").strip()
|
||||||
|
if not ts and not fl:
|
||||||
|
return None
|
||||||
|
v = cn_volume_int(ts)
|
||||||
|
if v > 0:
|
||||||
|
return v
|
||||||
|
v2 = cn_volume_int(fl)
|
||||||
|
if v2 > 0:
|
||||||
|
return v2
|
||||||
|
return 0
|
||||||
|
|
||||||
|
|
||||||
|
def comment_count_sort_value_from_cell(comment_count: str) -> int | None:
|
||||||
|
if not (comment_count or "").strip():
|
||||||
|
return None
|
||||||
|
return cn_volume_int(comment_count)
|
||||||
|
|
||||||
|
|
||||||
|
def comment_count_sort_value_from_merged(pipeline_comment_count: str) -> int | None:
|
||||||
|
"""宽表评价量列(与搜索侧口径类似的文案)。"""
|
||||||
|
return comment_count_sort_value_from_cell(pipeline_comment_count)
|
||||||
@ -27,6 +27,8 @@ const SORT_LABELS = {
|
|||||||
matrix_group_label: '类目',
|
matrix_group_label: '类目',
|
||||||
detail_category_path: '类目路径',
|
detail_category_path: '类目路径',
|
||||||
detail_brand: '品牌',
|
detail_brand: '品牌',
|
||||||
|
total_sales: '销量(解析排序)',
|
||||||
|
comment_count: '评价量(解析排序)',
|
||||||
}
|
}
|
||||||
|
|
||||||
const tab = ref('search')
|
const tab = ref('search')
|
||||||
|
|||||||
Loading…
x
Reference in New Issue
Block a user