From 9935a07279de89c0809759a6db6863f07da0f0ed Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=E7=BF=9F=E6=8C=81=E6=B1=9F?= <129171955+2513502304@users.noreply.github.com> Date: Sat, 19 Apr 2025 02:18:52 +0800 Subject: [PATCH 1/4] Add files via upload --- m_bilibili.py | 11 +++++++++++ 1 file changed, 11 insertions(+) create mode 100644 m_bilibili.py diff --git a/m_bilibili.py b/m_bilibili.py new file mode 100644 index 0000000..cddd1f1 --- /dev/null +++ b/m_bilibili.py @@ -0,0 +1,11 @@ +# 声明:本代码仅供学习和研究目的使用。使用者应遵守以下原则: +# 1. 不得用于任何商业用途。 +# 2. 使用时应遵守目标平台的使用条款和robots.txt规则。 +# 3. 不得进行大规模爬取或对平台造成运营干扰。 +# 4. 应合理控制请求频率,避免给目标平台带来不必要的负担。 +# 5. 不得用于任何非法或不当的用途。 +# +# 详细许可条款请参阅项目根目录下的LICENSE文件。 +# 使用本代码即表示您同意遵守上述原则和LICENSE中的所有条款。 + +# -*- coding: utf-8 -*- From ec97001451eddfbd788b8239c3e7d9cec0cb3104 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=E7=BF=9F=E6=8C=81=E6=B1=9F?= <129171955+2513502304@users.noreply.github.com> Date: Sat, 19 Apr 2025 02:22:22 +0800 Subject: [PATCH 2/4] Update tables.sql --- schema/tables.sql | 10 +++++++++- 1 file changed, 9 insertions(+), 1 deletion(-) diff --git a/schema/tables.sql b/schema/tables.sql index 3d0e7c6..479432d 100644 --- a/schema/tables.sql +++ b/schema/tables.sql @@ -16,7 +16,11 @@ CREATE TABLE `bilibili_video` `desc` longtext COMMENT '视频描述', `create_time` bigint NOT NULL COMMENT '视频发布时间戳', `liked_count` varchar(16) DEFAULT NULL COMMENT '视频点赞数', + `disliked_count` varchar(16) DEFAULT NULL COMMENT '视频点踩数', `video_play_count` varchar(16) DEFAULT NULL COMMENT '视频播放数量', + `video_favorite_count` varchar(16) DEFAULT NULL COMMENT '视频收藏数量', + `video_share_count` varchar(16) DEFAULT NULL COMMENT '视频分享数量', + `video_coin_count` varchar(16) DEFAULT NULL COMMENT '视频投币数量', `video_danmaku` varchar(16) DEFAULT NULL COMMENT '视频弹幕数量', `video_comment` varchar(16) DEFAULT NULL COMMENT '视频评论数量', `video_url` varchar(512) DEFAULT NULL COMMENT '视频详情URL', @@ -35,6 +39,8 @@ CREATE TABLE `bilibili_video_comment` `id` int NOT NULL AUTO_INCREMENT COMMENT '自增ID', `user_id` varchar(64) DEFAULT NULL COMMENT '用户ID', `nickname` varchar(64) DEFAULT NULL COMMENT '用户昵称', + `sex` varchar(64) DEFAULT NULL COMMENT '用户性别', + `sign` varchar(64) DEFAULT NULL COMMENT '用户签名', `avatar` varchar(255) DEFAULT NULL COMMENT '用户头像地址', `add_ts` bigint NOT NULL COMMENT '记录添加时间戳', `last_modify_ts` bigint NOT NULL COMMENT '记录最后修改时间戳', @@ -57,6 +63,8 @@ CREATE TABLE `bilibili_up_info` `id` int NOT NULL AUTO_INCREMENT COMMENT '自增ID', `user_id` varchar(64) DEFAULT NULL COMMENT '用户ID', `nickname` varchar(64) DEFAULT NULL COMMENT '用户昵称', + `sex` varchar(64) DEFAULT NULL COMMENT '用户性别', + `sign` varchar(64) DEFAULT NULL COMMENT '用户签名', `avatar` varchar(255) DEFAULT NULL COMMENT '用户头像地址', `add_ts` bigint NOT NULL COMMENT '记录添加时间戳', `last_modify_ts` bigint NOT NULL COMMENT '记录最后修改时间戳', @@ -537,4 +545,4 @@ CREATE TABLE `zhihu_creator` ( alter table douyin_aweme_comment add column `like_count` varchar(255) NOT NULL DEFAULT '0' COMMENT '点赞数'; alter table xhs_note add column xsec_token varchar(50) default null comment '签名算法'; -alter table douyin_aweme_comment add column `pictures` varchar(500) NOT NULL DEFAULT '' COMMENT '评论图片列表'; \ No newline at end of file +alter table douyin_aweme_comment add column `pictures` varchar(500) NOT NULL DEFAULT '' COMMENT '评论图片列表'; From b675547aabc32ebf82ced0dc6f100426cfb472bf Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=E7=BF=9F=E6=8C=81=E6=B1=9F?= <129171955+2513502304@users.noreply.github.com> Date: Sat, 19 Apr 2025 02:29:22 +0800 Subject: [PATCH 3/4] =?UTF-8?q?Update=20=5F=5Finit=5F=5F.py=EF=BC=8C?= =?UTF-8?q?=E4=B8=BAbilibili=E7=9A=84=E8=A7=86=E9=A2=91=E4=BF=A1=E6=81=AF?= =?UTF-8?q?=E3=80=81up=E4=B8=BB=E4=BF=A1=E6=81=AF=E3=80=81=E8=AF=84?= =?UTF-8?q?=E8=AE=BA=E4=BF=A1=E6=81=AF=E6=B7=BB=E5=8A=A0=E9=A2=9D=E5=A4=96?= =?UTF-8?q?=E5=AD=97=E6=AE=B5?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- store/bilibili/__init__.py | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/store/bilibili/__init__.py b/store/bilibili/__init__.py index 0d60a73..dcffa88 100644 --- a/store/bilibili/__init__.py +++ b/store/bilibili/__init__.py @@ -54,7 +54,11 @@ async def update_bilibili_video(video_item: Dict): "nickname": video_user_info.get("name"), "avatar": video_user_info.get("face", ""), "liked_count": str(video_item_stat.get("like", "")), + "disliked_count": str(video_item_stat.get("dislike", "")), "video_play_count": str(video_item_stat.get("view", "")), + "video_favorite_count": str(video_item_stat.get("favorite", "")), + "video_share_count": str(video_item_stat.get("share", "")), + "video_coin_count": str(video_item_stat.get("coin", "")), "video_danmaku": str(video_item_stat.get("danmaku", "")), "video_comment": str(video_item_stat.get("reply", "")), "last_modify_ts": utils.get_current_timestamp(), @@ -73,6 +77,8 @@ async def update_up_info(video_item: Dict): saver_up_info = { "user_id": str(video_item_card.get("mid")), "nickname": video_item_card.get("name"), + "sex": video_item_card.get("sex"), + "sign": video_item_card.get("sign"), "avatar": video_item_card.get("face"), "last_modify_ts": utils.get_current_timestamp(), "total_fans": video_item_card.get("fans"), @@ -105,6 +111,8 @@ async def update_bilibili_video_comment(video_id: str, comment_item: Dict): "content": content.get("message"), "user_id": user_info.get("mid"), "nickname": user_info.get("uname"), + "sex": user_info.get("sex"), + "sign": user_info.get("sign"), "avatar": user_info.get("avatar"), "sub_comment_count": str(comment_item.get("rcount", 0)), "last_modify_ts": utils.get_current_timestamp(), From af5a393a7afe4a8b8c1c4fa931d0671b7da524eb Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=E7=BF=9F=E6=8C=81=E6=B1=9F?= <129171955+2513502304@users.noreply.github.com> Date: Sat, 19 Apr 2025 04:34:24 +0800 Subject: [PATCH 4/4] =?UTF-8?q?Update=20core.py=EF=BC=8C=E5=88=A0=E9=99=A4?= =?UTF-8?q?=E4=BA=86=E5=85=B6=E5=AE=83=E4=BB=A3=E7=A0=81=E8=B4=A1=E7=8C=AE?= =?UTF-8?q?=E8=80=85=E6=89=80=E6=B7=BB=E5=8A=A0=E7=9A=84try-catch=E8=AF=AD?= =?UTF-8?q?=E5=8F=A5=EF=BC=8C=E8=AF=A5=E6=AE=B5try-catch=E8=AF=AD=E5=8F=A5?= =?UTF-8?q?=E5=B0=86=E4=BC=9A=E5=BD=B1=E5=93=8D=E5=85=B6=E4=BB=A3=E7=A0=81?= =?UTF-8?q?=E7=9A=84=E6=9C=80=E7=BB=88=E9=80=BB=E8=BE=91=E5=B9=B6=E4=BB=A4?= =?UTF-8?q?=E5=85=B6=E5=A4=B1=E6=95=88=EF=BC=8C=E4=BD=BF=E5=85=B6=E4=BB=85?= =?UTF-8?q?=E8=83=BD=E7=88=AC=E5=8F=96=E5=BD=93=E5=A4=A9=E4=B8=80=E5=A4=A9?= =?UTF-8?q?=E6=95=B0=E6=8D=AE=E8=80=8C=E6=97=A0=E6=B3=95=E8=B7=B3=E8=BD=AC?= =?UTF-8?q?=E5=88=B0=E4=B8=8B=E4=B8=80=E5=A4=A9=EF=BC=88=E5=8E=9F=E5=85=88?= =?UTF-8?q?=E7=9A=84=E9=80=BB=E8=BE=91=E5=B0=B1=E6=98=AFtry-catch=E6=8D=95?= =?UTF-8?q?=E8=8E=B7=E5=BC=82=E5=B8=B8=E4=BB=8E=E8=80=8C=E8=BF=9B=E5=85=A5?= =?UTF-8?q?=E4=B8=8B=E4=B8=80=E5=A4=A9=EF=BC=8C=E4=B8=8D=E8=A6=81=E5=86=8D?= =?UTF-8?q?=E5=90=91=E8=AF=A5=E8=AF=AD=E5=8F=A5=E4=B8=AD=E6=B7=BB=E5=8A=A0?= =?UTF-8?q?=E6=8D=95=E8=8E=B7=E5=BC=82=E5=B8=B8=E6=93=8D=E4=BD=9C=E6=88=96?= =?UTF-8?q?=E8=80=85finally=E8=AF=AD=E5=8F=A5=EF=BC=81=EF=BC=89?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- media_platform/bilibili/core.py | 21 +++++++++------------ 1 file changed, 9 insertions(+), 12 deletions(-) diff --git a/media_platform/bilibili/core.py b/media_platform/bilibili/core.py index 649fe19..5c7949a 100644 --- a/media_platform/bilibili/core.py +++ b/media_platform/bilibili/core.py @@ -164,10 +164,8 @@ class BilibiliCrawler(AbstractCrawler): task_list = [] try: task_list = [self.get_video_info_task(aid=video_item.get("aid"), bvid="", semaphore=semaphore) for video_item in video_list] - except Exception as e : - utils.logger.warning( - f"[BilibiliCrawler.search] error in the task list. The video for this page will not be included. {e}" - ) + except Exception as e: + utils.logger.warning(f"[BilibiliCrawler.search] error in the task list. The video for this page will not be included. {e}") video_items = await asyncio.gather(*task_list) for video_item in video_items: if video_item: @@ -177,16 +175,19 @@ class BilibiliCrawler(AbstractCrawler): await self.get_bilibili_video(video_item, semaphore) page += 1 await self.batch_get_video_comments(video_id_list) - # 按照 START_DAY 至 END_DAY 按照每一天进行筛选,这样能够突破 1000 条视频的限制,最大程度爬取该关键词下的所有视频 + # 按照 START_DAY 至 END_DAY 按照每一天进行筛选,这样能够突破 1000 条视频的限制,最大程度爬取该关键词下每一天的所有视频 else: for day in pd.date_range(start=config.START_DAY, end=config.END_DAY, freq='D'): # 按照每一天进行爬取的时间戳参数 pubtime_begin_s, pubtime_end_s = await self.get_pubtime_datetime(start=day.strftime('%Y-%m-%d'), end=day.strftime('%Y-%m-%d')) page = 1 + #!该段 while 语句在发生异常时(通常情况下为当天数据为空时)会自动跳转到下一天,以实现最大程度爬取该关键词下当天的所有视频 + #!除了仅保留现在原有的 try, except Exception 语句外,不要再添加其他的异常处理!!!否则将使该段代码失效,使其仅能爬取当天一天数据而无法跳转到下一天 + #!除非将该段代码的逻辑进行重构以实现相同的功能,否则不要进行修改!!! while (page - start_page + 1) * bili_limit_count <= config.CRAWLER_MAX_NOTES_COUNT: - # ! Catch any error if response return nothing, go to next day + #! Catch any error if response return nothing, go to next day try: - # ! Don't skip any page, to make sure gather all video in one day + #! Don't skip any page, to make sure gather all video in one day # if page < start_page: # utils.logger.info(f"[BilibiliCrawler.search] Skip page: {page}") # page += 1 @@ -205,11 +206,7 @@ class BilibiliCrawler(AbstractCrawler): video_list: List[Dict] = videos_res.get("result") semaphore = asyncio.Semaphore(config.MAX_CONCURRENCY_NUM) - task_list = [] - try: - task_list = [self.get_video_info_task(aid=video_item.get("aid"), bvid="", semaphore=semaphore) for video_item in video_list] - finally: - pass + task_list = [self.get_video_info_task(aid=video_item.get("aid"), bvid="", semaphore=semaphore) for video_item in video_list] video_items = await asyncio.gather(*task_list) for video_item in video_items: if video_item: