mirror of
https://github.com/RYDE-WORK/MediaCrawler.git
synced 2026-02-05 16:36:44 +08:00
commit
2a41b684ad
11
m_bilibili.py
Normal file
11
m_bilibili.py
Normal file
@ -0,0 +1,11 @@
|
|||||||
|
# 声明:本代码仅供学习和研究目的使用。使用者应遵守以下原则:
|
||||||
|
# 1. 不得用于任何商业用途。
|
||||||
|
# 2. 使用时应遵守目标平台的使用条款和robots.txt规则。
|
||||||
|
# 3. 不得进行大规模爬取或对平台造成运营干扰。
|
||||||
|
# 4. 应合理控制请求频率,避免给目标平台带来不必要的负担。
|
||||||
|
# 5. 不得用于任何非法或不当的用途。
|
||||||
|
#
|
||||||
|
# 详细许可条款请参阅项目根目录下的LICENSE文件。
|
||||||
|
# 使用本代码即表示您同意遵守上述原则和LICENSE中的所有条款。
|
||||||
|
|
||||||
|
# -*- coding: utf-8 -*-
|
||||||
@ -164,10 +164,8 @@ class BilibiliCrawler(AbstractCrawler):
|
|||||||
task_list = []
|
task_list = []
|
||||||
try:
|
try:
|
||||||
task_list = [self.get_video_info_task(aid=video_item.get("aid"), bvid="", semaphore=semaphore) for video_item in video_list]
|
task_list = [self.get_video_info_task(aid=video_item.get("aid"), bvid="", semaphore=semaphore) for video_item in video_list]
|
||||||
except Exception as e :
|
except Exception as e:
|
||||||
utils.logger.warning(
|
utils.logger.warning(f"[BilibiliCrawler.search] error in the task list. The video for this page will not be included. {e}")
|
||||||
f"[BilibiliCrawler.search] error in the task list. The video for this page will not be included. {e}"
|
|
||||||
)
|
|
||||||
video_items = await asyncio.gather(*task_list)
|
video_items = await asyncio.gather(*task_list)
|
||||||
for video_item in video_items:
|
for video_item in video_items:
|
||||||
if video_item:
|
if video_item:
|
||||||
@ -177,16 +175,19 @@ class BilibiliCrawler(AbstractCrawler):
|
|||||||
await self.get_bilibili_video(video_item, semaphore)
|
await self.get_bilibili_video(video_item, semaphore)
|
||||||
page += 1
|
page += 1
|
||||||
await self.batch_get_video_comments(video_id_list)
|
await self.batch_get_video_comments(video_id_list)
|
||||||
# 按照 START_DAY 至 END_DAY 按照每一天进行筛选,这样能够突破 1000 条视频的限制,最大程度爬取该关键词下的所有视频
|
# 按照 START_DAY 至 END_DAY 按照每一天进行筛选,这样能够突破 1000 条视频的限制,最大程度爬取该关键词下每一天的所有视频
|
||||||
else:
|
else:
|
||||||
for day in pd.date_range(start=config.START_DAY, end=config.END_DAY, freq='D'):
|
for day in pd.date_range(start=config.START_DAY, end=config.END_DAY, freq='D'):
|
||||||
# 按照每一天进行爬取的时间戳参数
|
# 按照每一天进行爬取的时间戳参数
|
||||||
pubtime_begin_s, pubtime_end_s = await self.get_pubtime_datetime(start=day.strftime('%Y-%m-%d'), end=day.strftime('%Y-%m-%d'))
|
pubtime_begin_s, pubtime_end_s = await self.get_pubtime_datetime(start=day.strftime('%Y-%m-%d'), end=day.strftime('%Y-%m-%d'))
|
||||||
page = 1
|
page = 1
|
||||||
|
#!该段 while 语句在发生异常时(通常情况下为当天数据为空时)会自动跳转到下一天,以实现最大程度爬取该关键词下当天的所有视频
|
||||||
|
#!除了仅保留现在原有的 try, except Exception 语句外,不要再添加其他的异常处理!!!否则将使该段代码失效,使其仅能爬取当天一天数据而无法跳转到下一天
|
||||||
|
#!除非将该段代码的逻辑进行重构以实现相同的功能,否则不要进行修改!!!
|
||||||
while (page - start_page + 1) * bili_limit_count <= config.CRAWLER_MAX_NOTES_COUNT:
|
while (page - start_page + 1) * bili_limit_count <= config.CRAWLER_MAX_NOTES_COUNT:
|
||||||
# ! Catch any error if response return nothing, go to next day
|
#! Catch any error if response return nothing, go to next day
|
||||||
try:
|
try:
|
||||||
# ! Don't skip any page, to make sure gather all video in one day
|
#! Don't skip any page, to make sure gather all video in one day
|
||||||
# if page < start_page:
|
# if page < start_page:
|
||||||
# utils.logger.info(f"[BilibiliCrawler.search] Skip page: {page}")
|
# utils.logger.info(f"[BilibiliCrawler.search] Skip page: {page}")
|
||||||
# page += 1
|
# page += 1
|
||||||
@ -205,11 +206,7 @@ class BilibiliCrawler(AbstractCrawler):
|
|||||||
video_list: List[Dict] = videos_res.get("result")
|
video_list: List[Dict] = videos_res.get("result")
|
||||||
|
|
||||||
semaphore = asyncio.Semaphore(config.MAX_CONCURRENCY_NUM)
|
semaphore = asyncio.Semaphore(config.MAX_CONCURRENCY_NUM)
|
||||||
task_list = []
|
task_list = [self.get_video_info_task(aid=video_item.get("aid"), bvid="", semaphore=semaphore) for video_item in video_list]
|
||||||
try:
|
|
||||||
task_list = [self.get_video_info_task(aid=video_item.get("aid"), bvid="", semaphore=semaphore) for video_item in video_list]
|
|
||||||
finally:
|
|
||||||
pass
|
|
||||||
video_items = await asyncio.gather(*task_list)
|
video_items = await asyncio.gather(*task_list)
|
||||||
for video_item in video_items:
|
for video_item in video_items:
|
||||||
if video_item:
|
if video_item:
|
||||||
|
|||||||
@ -16,7 +16,11 @@ CREATE TABLE `bilibili_video`
|
|||||||
`desc` longtext COMMENT '视频描述',
|
`desc` longtext COMMENT '视频描述',
|
||||||
`create_time` bigint NOT NULL COMMENT '视频发布时间戳',
|
`create_time` bigint NOT NULL COMMENT '视频发布时间戳',
|
||||||
`liked_count` varchar(16) DEFAULT NULL COMMENT '视频点赞数',
|
`liked_count` varchar(16) DEFAULT NULL COMMENT '视频点赞数',
|
||||||
|
`disliked_count` varchar(16) DEFAULT NULL COMMENT '视频点踩数',
|
||||||
`video_play_count` varchar(16) DEFAULT NULL COMMENT '视频播放数量',
|
`video_play_count` varchar(16) DEFAULT NULL COMMENT '视频播放数量',
|
||||||
|
`video_favorite_count` varchar(16) DEFAULT NULL COMMENT '视频收藏数量',
|
||||||
|
`video_share_count` varchar(16) DEFAULT NULL COMMENT '视频分享数量',
|
||||||
|
`video_coin_count` varchar(16) DEFAULT NULL COMMENT '视频投币数量',
|
||||||
`video_danmaku` varchar(16) DEFAULT NULL COMMENT '视频弹幕数量',
|
`video_danmaku` varchar(16) DEFAULT NULL COMMENT '视频弹幕数量',
|
||||||
`video_comment` varchar(16) DEFAULT NULL COMMENT '视频评论数量',
|
`video_comment` varchar(16) DEFAULT NULL COMMENT '视频评论数量',
|
||||||
`video_url` varchar(512) DEFAULT NULL COMMENT '视频详情URL',
|
`video_url` varchar(512) DEFAULT NULL COMMENT '视频详情URL',
|
||||||
@ -35,6 +39,8 @@ CREATE TABLE `bilibili_video_comment`
|
|||||||
`id` int NOT NULL AUTO_INCREMENT COMMENT '自增ID',
|
`id` int NOT NULL AUTO_INCREMENT COMMENT '自增ID',
|
||||||
`user_id` varchar(64) DEFAULT NULL COMMENT '用户ID',
|
`user_id` varchar(64) DEFAULT NULL COMMENT '用户ID',
|
||||||
`nickname` varchar(64) DEFAULT NULL COMMENT '用户昵称',
|
`nickname` varchar(64) DEFAULT NULL COMMENT '用户昵称',
|
||||||
|
`sex` varchar(64) DEFAULT NULL COMMENT '用户性别',
|
||||||
|
`sign` varchar(64) DEFAULT NULL COMMENT '用户签名',
|
||||||
`avatar` varchar(255) DEFAULT NULL COMMENT '用户头像地址',
|
`avatar` varchar(255) DEFAULT NULL COMMENT '用户头像地址',
|
||||||
`add_ts` bigint NOT NULL COMMENT '记录添加时间戳',
|
`add_ts` bigint NOT NULL COMMENT '记录添加时间戳',
|
||||||
`last_modify_ts` bigint NOT NULL COMMENT '记录最后修改时间戳',
|
`last_modify_ts` bigint NOT NULL COMMENT '记录最后修改时间戳',
|
||||||
@ -57,6 +63,8 @@ CREATE TABLE `bilibili_up_info`
|
|||||||
`id` int NOT NULL AUTO_INCREMENT COMMENT '自增ID',
|
`id` int NOT NULL AUTO_INCREMENT COMMENT '自增ID',
|
||||||
`user_id` varchar(64) DEFAULT NULL COMMENT '用户ID',
|
`user_id` varchar(64) DEFAULT NULL COMMENT '用户ID',
|
||||||
`nickname` varchar(64) DEFAULT NULL COMMENT '用户昵称',
|
`nickname` varchar(64) DEFAULT NULL COMMENT '用户昵称',
|
||||||
|
`sex` varchar(64) DEFAULT NULL COMMENT '用户性别',
|
||||||
|
`sign` varchar(64) DEFAULT NULL COMMENT '用户签名',
|
||||||
`avatar` varchar(255) DEFAULT NULL COMMENT '用户头像地址',
|
`avatar` varchar(255) DEFAULT NULL COMMENT '用户头像地址',
|
||||||
`add_ts` bigint NOT NULL COMMENT '记录添加时间戳',
|
`add_ts` bigint NOT NULL COMMENT '记录添加时间戳',
|
||||||
`last_modify_ts` bigint NOT NULL COMMENT '记录最后修改时间戳',
|
`last_modify_ts` bigint NOT NULL COMMENT '记录最后修改时间戳',
|
||||||
@ -537,4 +545,4 @@ CREATE TABLE `zhihu_creator` (
|
|||||||
alter table douyin_aweme_comment add column `like_count` varchar(255) NOT NULL DEFAULT '0' COMMENT '点赞数';
|
alter table douyin_aweme_comment add column `like_count` varchar(255) NOT NULL DEFAULT '0' COMMENT '点赞数';
|
||||||
|
|
||||||
alter table xhs_note add column xsec_token varchar(50) default null comment '签名算法';
|
alter table xhs_note add column xsec_token varchar(50) default null comment '签名算法';
|
||||||
alter table douyin_aweme_comment add column `pictures` varchar(500) NOT NULL DEFAULT '' COMMENT '评论图片列表';
|
alter table douyin_aweme_comment add column `pictures` varchar(500) NOT NULL DEFAULT '' COMMENT '评论图片列表';
|
||||||
|
|||||||
@ -54,7 +54,11 @@ async def update_bilibili_video(video_item: Dict):
|
|||||||
"nickname": video_user_info.get("name"),
|
"nickname": video_user_info.get("name"),
|
||||||
"avatar": video_user_info.get("face", ""),
|
"avatar": video_user_info.get("face", ""),
|
||||||
"liked_count": str(video_item_stat.get("like", "")),
|
"liked_count": str(video_item_stat.get("like", "")),
|
||||||
|
"disliked_count": str(video_item_stat.get("dislike", "")),
|
||||||
"video_play_count": str(video_item_stat.get("view", "")),
|
"video_play_count": str(video_item_stat.get("view", "")),
|
||||||
|
"video_favorite_count": str(video_item_stat.get("favorite", "")),
|
||||||
|
"video_share_count": str(video_item_stat.get("share", "")),
|
||||||
|
"video_coin_count": str(video_item_stat.get("coin", "")),
|
||||||
"video_danmaku": str(video_item_stat.get("danmaku", "")),
|
"video_danmaku": str(video_item_stat.get("danmaku", "")),
|
||||||
"video_comment": str(video_item_stat.get("reply", "")),
|
"video_comment": str(video_item_stat.get("reply", "")),
|
||||||
"last_modify_ts": utils.get_current_timestamp(),
|
"last_modify_ts": utils.get_current_timestamp(),
|
||||||
@ -73,6 +77,8 @@ async def update_up_info(video_item: Dict):
|
|||||||
saver_up_info = {
|
saver_up_info = {
|
||||||
"user_id": str(video_item_card.get("mid")),
|
"user_id": str(video_item_card.get("mid")),
|
||||||
"nickname": video_item_card.get("name"),
|
"nickname": video_item_card.get("name"),
|
||||||
|
"sex": video_item_card.get("sex"),
|
||||||
|
"sign": video_item_card.get("sign"),
|
||||||
"avatar": video_item_card.get("face"),
|
"avatar": video_item_card.get("face"),
|
||||||
"last_modify_ts": utils.get_current_timestamp(),
|
"last_modify_ts": utils.get_current_timestamp(),
|
||||||
"total_fans": video_item_card.get("fans"),
|
"total_fans": video_item_card.get("fans"),
|
||||||
@ -105,6 +111,8 @@ async def update_bilibili_video_comment(video_id: str, comment_item: Dict):
|
|||||||
"content": content.get("message"),
|
"content": content.get("message"),
|
||||||
"user_id": user_info.get("mid"),
|
"user_id": user_info.get("mid"),
|
||||||
"nickname": user_info.get("uname"),
|
"nickname": user_info.get("uname"),
|
||||||
|
"sex": user_info.get("sex"),
|
||||||
|
"sign": user_info.get("sign"),
|
||||||
"avatar": user_info.get("avatar"),
|
"avatar": user_info.get("avatar"),
|
||||||
"sub_comment_count": str(comment_item.get("rcount", 0)),
|
"sub_comment_count": str(comment_item.get("rcount", 0)),
|
||||||
"last_modify_ts": utils.get_current_timestamp(),
|
"last_modify_ts": utils.get_current_timestamp(),
|
||||||
|
|||||||
Loading…
x
Reference in New Issue
Block a user