refactor(jd_pc_search): 拆分搜索与评论的解析/落盘与请求脚本

评论:新增 jd_item_comment_parse、jd_item_comment_export;jd_h5_item_comment_requests 仅保留 Node 与 Playwright 与 main。搜索:新增 jd_h5_search_parse;jd_h5_search_requests 精简并 re-export CSV_FIELDS 等以兼容流水线与 collect_pc_search_items。

Made-with: Cursor
This commit is contained in:
hub-gif 2026-04-17 14:06:14 +08:00
parent 1431ed3204
commit 179c9b2659
5 changed files with 1581 additions and 1493 deletions

View File

@ -12,17 +12,18 @@
用法本仓库默认: 修改下方运行配置 ``python jd_h5_item_comment_requests.py``无命令行参数
**模块划分**评价 JSON 解析见 ``jd_item_comment_parse.py``CSV/JSONL 落盘见 ``jd_item_comment_export.py``
本文件保留 Node 签请求Playwright 采集与 ``main`` re-export 解析/导出符号以兼容 ``jd_keyword_pipeline`` 等导入
"""
from __future__ import annotations
import csv
import json
import random
import subprocess
import sys
import time
from io import StringIO
from pathlib import Path
from types import SimpleNamespace
from typing import Any
@ -32,14 +33,27 @@ from playwright.sync_api import sync_playwright
_JD_PKG_ROOT = Path(__file__).resolve().parent.parent
if str(_JD_PKG_ROOT) not in sys.path:
sys.path.insert(0, str(_JD_PKG_ROOT))
_JD_COMMENT_DIR = Path(__file__).resolve().parent
if str(_JD_COMMENT_DIR) not in sys.path:
sys.path.insert(0, str(_JD_COMMENT_DIR))
_BACKEND_ROOT = Path(__file__).resolve().parents[3]
if str(_BACKEND_ROOT) not in sys.path:
sys.path.insert(0, str(_BACKEND_ROOT))
from common.jd_delay_utils import parse_request_delay_range
from pipeline.csv_schema import COMMENT_CSV_COLUMNS, COMMENT_ROW_DICT_KEYS # noqa: E402
from _low_gi_root import low_gi_project_root # noqa: E402
_JD_COMMENT_DIR = Path(__file__).resolve().parent
from jd_item_comment_export import ( # noqa: E402
append_comments_jsonl,
write_comments_flat_csv,
)
from jd_item_comment_parse import ( # noqa: E402
category_and_first_guid_from_lego,
extract_comment_rows_from_jsonl_file,
extract_comment_rows_from_parsed,
jd_business_ok,
loads_jd_plain_json,
parse_list_pages_spec,
)
# ---------------------------------------------------------------------------
# 运行配置(按需改这里)
@ -193,40 +207,6 @@ def _sleep_between_jd_requests(
time.sleep(sec)
def parse_list_pages_spec(spec: str) -> list[str]:
"""
--list-pages``1-5`` 1..5``1,3,5`` 单页序列单数字 ``2`` [\"2\"]。
"""
s = (spec or "").strip()
if not s:
return ["1"]
if "," in s:
return [p.strip() for p in s.split(",") if p.strip()]
if "-" in s:
parts = s.split("-", 1)
lo, hi = int(parts[0].strip()), int(parts[1].strip())
if lo > hi:
lo, hi = hi, lo
return [str(i) for i in range(lo, hi + 1)]
return [s]
def category_and_first_guid_from_lego(parsed: Any) -> tuple[str, str]:
"""从 getLegoWareDetailComment 的 commentInfoList[0] 取 categorymaidianInfo 前缀)与 guid。"""
if not isinstance(parsed, dict):
return "", ""
lst = parsed.get("commentInfoList")
if not isinstance(lst, list) or not lst:
return "", ""
first = lst[0]
if not isinstance(first, dict):
return "", ""
guid = str(first.get("guid") or "").strip()
maidian = str(first.get("maidianInfo") or "").strip()
category = maidian.split("_", 1)[0].strip() if maidian else ""
return category, guid
def _read_sku_lines(path: str) -> list[str]:
p = Path(path)
if not p.is_file():
@ -241,162 +221,6 @@ def _read_sku_lines(path: str) -> list[str]:
return out
def _loads_jd_plain_json(text: str) -> Any:
s = (text or "").strip()
if not s:
return None
try:
return json.loads(s)
except json.JSONDecodeError:
return None
def _jd_business_ok(parsed: Any) -> bool:
if not isinstance(parsed, dict):
return False
if parsed.get("success") is False:
return False
c = parsed.get("code")
if c is None:
return True
return c == 0 or str(c) == "0"
def _clean_text(v: Any) -> str:
if v is None:
return ""
s = str(v).strip()
return " ".join(s.split()) if s else ""
def _large_pic_urls_from_picture_list(pil: Any) -> list[str]:
out: list[str] = []
if not isinstance(pil, list):
return out
for p in pil:
if not isinstance(p, dict):
continue
u = p.get("largePicURL") or p.get("largePicUrl")
if u:
t = str(u).strip()
if t and t not in out:
out.append(t)
return out
def _is_jd_single_comment_dict(d: dict) -> bool:
"""区分「一条评价」与标签/楼层等对象getCommentListPage 里多为 commentInfo 扁平结构)。"""
cid = d.get("commentId")
if cid is None or str(cid).strip() == "":
return False
if d.get("userNickName") is None and not (
d.get("tagCommentContent") or d.get("commentData")
):
return False
return True
def _walk_collect_comment_dicts(obj: Any, acc: list[dict[str, Any]]) -> None:
"""深度遍历 JSON收集所有像单条评价的 dict含 Lego 的 commentInfoList 项与列表页的 commentInfo"""
if isinstance(obj, dict):
if _is_jd_single_comment_dict(obj):
acc.append(obj)
for v in obj.values():
_walk_collect_comment_dicts(v, acc)
elif isinstance(obj, list):
for x in obj:
_walk_collect_comment_dicts(x, acc)
def _row_from_comment_dict(sku: str, item: dict[str, Any]) -> dict[str, Any]:
text = _clean_text(
item.get("tagCommentContent") or item.get("commentData")
)
buy = _clean_text(
item.get("buyCountText") or item.get("repurchaseInfo")
)
date = _clean_text(
item.get("commentDate") or item.get("newCommentDate")
)
return {
"sku": str(sku).strip(),
"commentId": str(item.get("commentId") or "").strip(),
"userNickName": _clean_text(item.get("userNickName")),
"tagCommentContent": text,
"commentDate": date,
"buyCountText": buy,
"largePicURLs": _large_pic_urls_from_picture_list(
item.get("pictureInfoList")
),
"commentScore":str(item.get("commentScore") or "").strip(),
}
def extract_comment_rows_from_parsed(sku: str, parsed: Any) -> list[dict[str, Any]]:
"""
从整段 parsed 深度遍历抽取评价
- getLegoWareDetailCommentcommentInfoList / lastCommentInfoList
- getCommentListPageresult.floors data { commentInfo: {...} } 已拍平为内层字段同上
"""
if not isinstance(parsed, dict):
return []
acc: list[dict[str, Any]] = []
_walk_collect_comment_dicts(parsed, acc)
seen: set[str] = set()
rows: list[dict[str, Any]] = []
for item in acc:
cid = str(item.get("commentId") or "").strip()
dedup_key = f"{sku}:{cid}" if cid else f"{sku}:{id(item)}"
if dedup_key in seen:
continue
seen.add(dedup_key)
rows.append(_row_from_comment_dict(sku, item))
return rows
def _write_comments_csv(path: Path, rows: list[dict[str, Any]]) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
buf = StringIO()
fn = list(COMMENT_CSV_COLUMNS)
w = csv.DictWriter(buf, fieldnames=fn, extrasaction="ignore")
w.writeheader()
for r in rows:
line: dict[str, Any] = {}
for h, api_k in zip(COMMENT_CSV_COLUMNS, COMMENT_ROW_DICT_KEYS):
if api_k == "largePicURLs":
line[h] = json.dumps(r.get("largePicURLs") or [], ensure_ascii=False)
else:
line[h] = r.get(api_k, "")
w.writerow(line)
path.write_text("\ufeff" + buf.getvalue(), encoding="utf-8")
def write_comments_flat_csv(path: Path | str, rows: list[dict[str, Any]]) -> None:
"""与 ``COMMENTS_OUT`` 为 ``.csv`` 时相同格式UTF-8 BOM供流水线等复用。"""
_write_comments_csv(Path(path), rows)
def _append_comments_jsonl(f, rows: list[dict[str, Any]]) -> None:
for r in rows:
f.write(json.dumps(r, ensure_ascii=False) + "\n")
def _extract_rows_from_jsonl_file(path: Path) -> list[dict[str, Any]]:
all_rows: list[dict[str, Any]] = []
for line in path.read_text(encoding="utf-8").splitlines():
s = line.strip()
if not s or s.startswith("#"):
continue
try:
rec = json.loads(s)
except json.JSONDecodeError:
continue
sku = str(rec.get("sku") or "").strip()
parsed = rec.get("parsed")
all_rows.extend(extract_comment_rows_from_parsed(sku, parsed))
return all_rows
def main() -> None:
args = SimpleNamespace(
sku=(SKU or "").strip(),
@ -429,14 +253,14 @@ def main() -> None:
print("离线模式:请配置 FROM_JSONL 与 COMMENTS_OUT", file=sys.stderr)
sys.exit(2)
co_path = Path(comments_out)
rows = _extract_rows_from_jsonl_file(Path(from_jsonl))
rows = extract_comment_rows_from_jsonl_file(Path(from_jsonl))
suf = co_path.suffix.lower()
if suf == ".csv":
_write_comments_csv(co_path, rows)
write_comments_flat_csv(co_path, rows)
elif suf == ".jsonl":
co_path.parent.mkdir(parents=True, exist_ok=True)
with co_path.open("w", encoding="utf-8") as cf:
_append_comments_jsonl(cf, rows)
append_comments_jsonl(cf, rows)
else:
print("COMMENTS_OUT 请使用 .csv 或 .jsonl 扩展名", file=sys.stderr)
sys.exit(2)
@ -519,13 +343,13 @@ def main() -> None:
print(f"[京东] HTTP {status}--raise-http 已启用", file=sys.stderr)
sys.exit(1)
parsed = _loads_jd_plain_json(text)
parsed = loads_jd_plain_json(text)
http_ok = 200 <= status < 300
row = {
"sku": sku,
"http_status": status,
"http_ok": http_ok,
"ok": http_ok and _jd_business_ok(parsed),
"ok": http_ok and jd_business_ok(parsed),
"parsed": parsed,
"raw": text if parsed is None else None,
}
@ -545,7 +369,7 @@ def main() -> None:
flat = extract_comment_rows_from_parsed(sku, parsed)
if comments_path is not None:
if comments_jsonl_f is not None:
_append_comments_jsonl(comments_jsonl_f, flat)
append_comments_jsonl(comments_jsonl_f, flat)
else:
comments_csv_rows.extend(flat)
@ -614,7 +438,7 @@ def main() -> None:
file=sys.stderr,
)
sys.exit(1)
parsed_p = _loads_jd_plain_json(text_p)
parsed_p = loads_jd_plain_json(text_p)
http_ok_p = 200 <= st_p < 300
row_p = {
"sku": sku,
@ -623,7 +447,7 @@ def main() -> None:
"http_status": st_p,
"http_ok": http_ok_p,
"ok": http_ok_p
and _jd_business_ok(parsed_p),
and jd_business_ok(parsed_p),
"parsed": parsed_p,
"raw": text_p if parsed_p is None else None,
}
@ -638,7 +462,7 @@ def main() -> None:
)
if comments_path is not None:
if comments_jsonl_f is not None:
_append_comments_jsonl(
append_comments_jsonl(
comments_jsonl_f, flat_p
)
else:
@ -652,7 +476,7 @@ def main() -> None:
if comments_jsonl_f:
comments_jsonl_f.close()
if comments_path is not None and comments_path.suffix.lower() == ".csv":
_write_comments_csv(comments_path, comments_csv_rows)
write_comments_flat_csv(comments_path, comments_csv_rows)
if __name__ == "__main__":

View File

@ -0,0 +1,46 @@
# -*- coding: utf-8 -*-
"""
评价扁平结果**落盘**UTF-8 BOM CSV / JSONL 行追加
解析见 ``jd_item_comment_parse``
"""
from __future__ import annotations
import csv
import json
import sys
from io import StringIO
from pathlib import Path
from typing import Any
_BACKEND_ROOT = Path(__file__).resolve().parents[3]
if str(_BACKEND_ROOT) not in sys.path:
sys.path.insert(0, str(_BACKEND_ROOT))
from pipeline.csv_schema import COMMENT_CSV_COLUMNS, COMMENT_ROW_DICT_KEYS # noqa: E402
def write_comments_flat_csv(path: Path | str, rows: list[dict[str, Any]]) -> None:
"""与 COMMENTS_OUT 为 ``.csv`` 时相同格式UTF-8 BOM供流水线等复用。"""
_write_comments_csv(Path(path), rows)
def _write_comments_csv(path: Path, rows: list[dict[str, Any]]) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
buf = StringIO()
fn = list(COMMENT_CSV_COLUMNS)
w = csv.DictWriter(buf, fieldnames=fn, extrasaction="ignore")
w.writeheader()
for r in rows:
line: dict[str, Any] = {}
for h, api_k in zip(COMMENT_CSV_COLUMNS, COMMENT_ROW_DICT_KEYS):
if api_k == "largePicURLs":
line[h] = json.dumps(r.get("largePicURLs") or [], ensure_ascii=False)
else:
line[h] = r.get(api_k, "")
w.writerow(line)
path.write_text("\ufeff" + buf.getvalue(), encoding="utf-8")
def append_comments_jsonl(f: Any, rows: list[dict[str, Any]]) -> None:
for r in rows:
f.write(json.dumps(r, ensure_ascii=False) + "\n")

View File

@ -0,0 +1,174 @@
# -*- coding: utf-8 -*-
"""
京东商品评价 JSON **纯解析** Lego / 列表页响应中抽取扁平评价行
请求签名与 Playwright ``jd_h5_item_comment_requests``
"""
from __future__ import annotations
import json
from pathlib import Path
from typing import Any
def parse_list_pages_spec(spec: str) -> list[str]:
"""
--list-pages``1-5`` 1..5``1,3,5`` 单页序列单数字 ``2`` [\"2\"]。
"""
s = (spec or "").strip()
if not s:
return ["1"]
if "," in s:
return [p.strip() for p in s.split(",") if p.strip()]
if "-" in s:
parts = s.split("-", 1)
lo, hi = int(parts[0].strip()), int(parts[1].strip())
if lo > hi:
lo, hi = hi, lo
return [str(i) for i in range(lo, hi + 1)]
return [s]
def category_and_first_guid_from_lego(parsed: Any) -> tuple[str, str]:
"""从 getLegoWareDetailComment 的 commentInfoList[0] 取 categorymaidianInfo 前缀)与 guid。"""
if not isinstance(parsed, dict):
return "", ""
lst = parsed.get("commentInfoList")
if not isinstance(lst, list) or not lst:
return "", ""
first = lst[0]
if not isinstance(first, dict):
return "", ""
guid = str(first.get("guid") or "").strip()
maidian = str(first.get("maidianInfo") or "").strip()
category = maidian.split("_", 1)[0].strip() if maidian else ""
return category, guid
def loads_jd_plain_json(text: str) -> Any:
s = (text or "").strip()
if not s:
return None
try:
return json.loads(s)
except json.JSONDecodeError:
return None
def jd_business_ok(parsed: Any) -> bool:
if not isinstance(parsed, dict):
return False
if parsed.get("success") is False:
return False
c = parsed.get("code")
if c is None:
return True
return c == 0 or str(c) == "0"
def _clean_text(v: Any) -> str:
if v is None:
return ""
s = str(v).strip()
return " ".join(s.split()) if s else ""
def _large_pic_urls_from_picture_list(pil: Any) -> list[str]:
out: list[str] = []
if not isinstance(pil, list):
return out
for p in pil:
if not isinstance(p, dict):
continue
u = p.get("largePicURL") or p.get("largePicUrl")
if u:
t = str(u).strip()
if t and t not in out:
out.append(t)
return out
def _is_jd_single_comment_dict(d: dict) -> bool:
"""区分「一条评价」与标签/楼层等对象getCommentListPage 里多为 commentInfo 扁平结构)。"""
cid = d.get("commentId")
if cid is None or str(cid).strip() == "":
return False
if d.get("userNickName") is None and not (
d.get("tagCommentContent") or d.get("commentData")
):
return False
return True
def _walk_collect_comment_dicts(obj: Any, acc: list[dict[str, Any]]) -> None:
"""深度遍历 JSON收集所有像单条评价的 dict含 Lego 的 commentInfoList 项与列表页的 commentInfo"""
if isinstance(obj, dict):
if _is_jd_single_comment_dict(obj):
acc.append(obj)
for v in obj.values():
_walk_collect_comment_dicts(v, acc)
elif isinstance(obj, list):
for x in obj:
_walk_collect_comment_dicts(x, acc)
def _row_from_comment_dict(sku: str, item: dict[str, Any]) -> dict[str, Any]:
text = _clean_text(
item.get("tagCommentContent") or item.get("commentData")
)
buy = _clean_text(
item.get("buyCountText") or item.get("repurchaseInfo")
)
date = _clean_text(
item.get("commentDate") or item.get("newCommentDate")
)
return {
"sku": str(sku).strip(),
"commentId": str(item.get("commentId") or "").strip(),
"userNickName": _clean_text(item.get("userNickName")),
"tagCommentContent": text,
"commentDate": date,
"buyCountText": buy,
"largePicURLs": _large_pic_urls_from_picture_list(
item.get("pictureInfoList")
),
"commentScore": str(item.get("commentScore") or "").strip(),
}
def extract_comment_rows_from_parsed(sku: str, parsed: Any) -> list[dict[str, Any]]:
"""
从整段 parsed 深度遍历抽取评价
- getLegoWareDetailCommentcommentInfoList / lastCommentInfoList
- getCommentListPageresult.floors data { commentInfo: {...} } 已拍平为内层字段同上
"""
if not isinstance(parsed, dict):
return []
acc: list[dict[str, Any]] = []
_walk_collect_comment_dicts(parsed, acc)
seen: set[str] = set()
rows: list[dict[str, Any]] = []
for item in acc:
cid = str(item.get("commentId") or "").strip()
dedup_key = f"{sku}:{cid}" if cid else f"{sku}:{id(item)}"
if dedup_key in seen:
continue
seen.add(dedup_key)
rows.append(_row_from_comment_dict(sku, item))
return rows
def extract_comment_rows_from_jsonl_file(path: Path) -> list[dict[str, Any]]:
"""离线:从采集 JSONL 每行 { sku, parsed } 抽取扁平评价行。"""
all_rows: list[dict[str, Any]] = []
for line in path.read_text(encoding="utf-8").splitlines():
s = line.strip()
if not s or s.startswith("#"):
continue
try:
rec = json.loads(s)
except json.JSONDecodeError:
continue
sku = str(rec.get("sku") or "").strip()
parsed = rec.get("parsed")
all_rows.extend(extract_comment_rows_from_parsed(sku, parsed))
return all_rows

File diff suppressed because it is too large Load Diff