i18n: translate all Chinese comments, docstrings, and logger messages to English

Comprehensive translation of Chinese text to English across the entire codebase:

- api/: FastAPI server documentation and logger messages
- cache/: Cache abstraction layer comments and docstrings
- database/: Database models and MongoDB store documentation
- media_platform/: All platform crawlers (Bilibili, Douyin, Kuaishou, Tieba, Weibo, Xiaohongshu, Zhihu)
- model/: Data model documentation
- proxy/: Proxy pool and provider documentation
- store/: Data storage layer comments
- tools/: Utility functions and browser automation
- test/: Test file documentation

Preserved: Chinese disclaimer header (lines 10-18) for legal compliance

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude Opus 4.5 <noreply@anthropic.com>
This commit is contained in:
程序员阿江(Relakkes)
2025-12-26 23:27:19 +08:00
co-authored by Claude Opus 4.5
parent 1544d13dd5
commit 157ddfb21b
93 changed files with 1971 additions and 1955 deletions
+9 -9
View File
@@ -21,7 +21,7 @@
# -*- coding: utf-8 -*-
# @Author : persist1@126.com
# @Time : 2025/9/5 19:34
# @Desc : B站存储实现类
# @Desc : Bilibili storage implementation class
import asyncio
import csv
import json
@@ -310,16 +310,16 @@ class BiliSqliteStoreImplement(BiliDbStoreImplement):
class BiliMongoStoreImplement(AbstractStore):
"""B站MongoDB存储实现"""
"""Bilibili MongoDB storage implementation"""
def __init__(self):
self.mongo_store = MongoDBStoreBase(collection_prefix="bilibili")
async def store_content(self, content_item: Dict):
"""
存储视频内容到MongoDB
Store video content to MongoDB
Args:
content_item: 视频内容数据
content_item: Video content data
"""
video_id = content_item.get("video_id")
if not video_id:
@@ -334,9 +334,9 @@ class BiliMongoStoreImplement(AbstractStore):
async def store_comment(self, comment_item: Dict):
"""
存储评论到MongoDB
Store comment to MongoDB
Args:
comment_item: 评论数据
comment_item: Comment data
"""
comment_id = comment_item.get("comment_id")
if not comment_id:
@@ -351,9 +351,9 @@ class BiliMongoStoreImplement(AbstractStore):
async def store_creator(self, creator_item: Dict):
"""
存储UP主信息到MongoDB
Store UP master information to MongoDB
Args:
creator_item: UP主数据
creator_item: UP master data
"""
user_id = creator_item.get("user_id")
if not user_id:
@@ -368,7 +368,7 @@ class BiliMongoStoreImplement(AbstractStore):
class BiliExcelStoreImplement:
"""B站Excel存储实现 - 全局单例"""
"""Bilibili Excel storage implementation - Global singleton"""
def __new__(cls, *args, **kwargs):
from store.excel_store_base import ExcelStoreBase
+1 -1
View File
@@ -20,7 +20,7 @@
# -*- coding: utf-8 -*-
# @Author : helloteemo
# @Time : 2024/7/12 20:01
# @Desc : bilibili 媒体保存
# @Desc : Bilibili media storage
import pathlib
from typing import Dict
+20 -20
View File
@@ -50,13 +50,13 @@ class DouyinStoreFactory:
def _extract_note_image_list(aweme_detail: Dict) -> List[str]:
"""
提取笔记图片列表
Extract note image list
Args:
aweme_detail (Dict): 抖音内容详情
aweme_detail (Dict): Douyin content details
Returns:
List[str]: 笔记图片列表
List[str]: Note image list
"""
images_res: List[str] = []
images: List[Dict] = aweme_detail.get("images", [])
@@ -65,7 +65,7 @@ def _extract_note_image_list(aweme_detail: Dict) -> List[str]:
return []
for image in images:
image_url_list = image.get("url_list", []) # download_url_list 为带水印的图片,url_list 为无水印的图片
image_url_list = image.get("url_list", []) # download_url_list has watermarked images, url_list has non-watermarked images
if image_url_list:
images_res.append(image_url_list[-1])
@@ -74,13 +74,13 @@ def _extract_note_image_list(aweme_detail: Dict) -> List[str]:
def _extract_comment_image_list(comment_item: Dict) -> List[str]:
"""
提取评论图片列表
Extract comment image list
Args:
comment_item (Dict): 抖音评论
comment_item (Dict): Douyin comment
Returns:
List[str]: 评论图片列表
List[str]: Comment image list
"""
images_res: List[str] = []
image_list: List[Dict] = comment_item.get("image_list", [])
@@ -98,13 +98,13 @@ def _extract_comment_image_list(comment_item: Dict) -> List[str]:
def _extract_content_cover_url(aweme_detail: Dict) -> str:
"""
提取视频封面地址
Extract video cover URL
Args:
aweme_detail (Dict): 抖音内容详情
aweme_detail (Dict): Douyin content details
Returns:
str: 视频封面地址
str: Video cover URL
"""
res_cover_url = ""
@@ -118,13 +118,13 @@ def _extract_content_cover_url(aweme_detail: Dict) -> str:
def _extract_video_download_url(aweme_detail: Dict) -> str:
"""
提取视频下载地址
Extract video download URL
Args:
aweme_detail (Dict): 抖音视频
aweme_detail (Dict): Douyin video
Returns:
str: 视频下载地址
str: Video download URL
"""
video_item = aweme_detail.get("video", {})
url_h264_list = video_item.get("play_addr_h264", {}).get("url_list", [])
@@ -138,13 +138,13 @@ def _extract_video_download_url(aweme_detail: Dict) -> str:
def _extract_music_download_url(aweme_detail: Dict) -> str:
"""
提取音乐下载地址
Extract music download URL
Args:
aweme_detail (Dict): 抖音视频
aweme_detail (Dict): Douyin video
Returns:
str: 音乐下载地址
str: Music download URL
"""
music_item = aweme_detail.get("music", {})
play_url = music_item.get("play_url", {})
@@ -228,12 +228,12 @@ async def update_dy_aweme_comment(aweme_id: str, comment_item: Dict):
async def save_creator(user_id: str, creator: Dict):
user_info = creator.get("user", {})
gender_map = {0: "未知", 1: "男", 2: "女"}
gender_map = {0: "Unknown", 1: "Male", 2: "Female"}
avatar_uri = user_info.get("avatar_300x300", {}).get("uri")
local_db_item = {
"user_id": user_id,
"nickname": user_info.get("nickname"),
"gender": gender_map.get(user_info.get("gender"), "未知"),
"gender": gender_map.get(user_info.get("gender"), "Unknown"),
"avatar": f"https://p3-pc.douyinpic.com/img/{avatar_uri}" + r"~c5_300x300.jpeg?from=2956013662",
"desc": user_info.get("signature"),
"ip_location": user_info.get("ip_location"),
@@ -249,7 +249,7 @@ async def save_creator(user_id: str, creator: Dict):
async def update_dy_aweme_image(aweme_id, pic_content, extension_file_name):
"""
更新抖音笔记图片
Update Douyin note image
Args:
aweme_id:
pic_content:
@@ -264,7 +264,7 @@ async def update_dy_aweme_image(aweme_id, pic_content, extension_file_name):
async def update_dy_aweme_video(aweme_id, video_content, extension_file_name):
"""
更新抖音短视频
Update Douyin short video
Args:
aweme_id:
video_content:
+9 -9
View File
@@ -21,7 +21,7 @@
# -*- coding: utf-8 -*-
# @Author : persist1@126.com
# @Time : 2025/9/5 19:34
# @Desc : 抖音存储实现类
# @Desc : Douyin storage implementation class
import asyncio
import json
import os
@@ -209,16 +209,16 @@ class DouyinSqliteStoreImplement(DouyinDbStoreImplement):
class DouyinMongoStoreImplement(AbstractStore):
"""抖音MongoDB存储实现"""
"""Douyin MongoDB storage implementation"""
def __init__(self):
self.mongo_store = MongoDBStoreBase(collection_prefix="douyin")
async def store_content(self, content_item: Dict):
"""
存储视频内容到MongoDB
Store video content to MongoDB
Args:
content_item: 视频内容数据
content_item: Video content data
"""
aweme_id = content_item.get("aweme_id")
if not aweme_id:
@@ -233,9 +233,9 @@ class DouyinMongoStoreImplement(AbstractStore):
async def store_comment(self, comment_item: Dict):
"""
存储评论到MongoDB
Store comment to MongoDB
Args:
comment_item: 评论数据
comment_item: Comment data
"""
comment_id = comment_item.get("comment_id")
if not comment_id:
@@ -250,9 +250,9 @@ class DouyinMongoStoreImplement(AbstractStore):
async def store_creator(self, creator_item: Dict):
"""
存储创作者信息到MongoDB
Store creator information to MongoDB
Args:
creator_item: 创作者数据
creator_item: Creator data
"""
user_id = creator_item.get("user_id")
if not user_id:
@@ -267,7 +267,7 @@ class DouyinMongoStoreImplement(AbstractStore):
class DouyinExcelStoreImplement:
"""抖音Excel存储实现 - 全局单例"""
"""Douyin Excel storage implementation - Global singleton"""
def __new__(cls, *args, **kwargs):
from store.excel_store_base import ExcelStoreBase
+1 -1
View File
@@ -109,7 +109,7 @@ async def save_creator(user_id: str, creator: Dict):
local_db_item = {
'user_id': user_id,
'nickname': profile.get('user_name'),
'gender': '女' if profile.get('gender') == "F" else '男',
'gender': 'Female' if profile.get('gender') == "F" else 'Male',
'avatar': profile.get('headurl'),
'desc': profile.get('user_text'),
'ip_location': "",
+10 -10
View File
@@ -21,7 +21,7 @@
# -*- coding: utf-8 -*-
# @Author : persist1@126.com
# @Time : 2025/9/5 19:34
# @Desc : 快手存储实现类
# @Desc : Kuaishou storage implementation class
import asyncio
import csv
import json
@@ -43,7 +43,7 @@ from database.mongodb_store_base import MongoDBStoreBase
def calculate_number_of_files(file_store_path: str) -> int:
"""计算数据保存文件的前部分排序数字,支持每次运行代码不写到同一个文件中
"""Calculate the prefix sorting number for data save files, supporting writing to different files for each run
Args:
file_store_path;
Returns:
@@ -171,16 +171,16 @@ class KuaishouSqliteStoreImplement(KuaishouDbStoreImplement):
class KuaishouMongoStoreImplement(AbstractStore):
"""快手MongoDB存储实现"""
"""Kuaishou MongoDB storage implementation"""
def __init__(self):
self.mongo_store = MongoDBStoreBase(collection_prefix="kuaishou")
async def store_content(self, content_item: Dict):
"""
存储视频内容到MongoDB
Store video content to MongoDB
Args:
content_item: 视频内容数据
content_item: Video content data
"""
video_id = content_item.get("video_id")
if not video_id:
@@ -195,9 +195,9 @@ class KuaishouMongoStoreImplement(AbstractStore):
async def store_comment(self, comment_item: Dict):
"""
存储评论到MongoDB
Store comment to MongoDB
Args:
comment_item: 评论数据
comment_item: Comment data
"""
comment_id = comment_item.get("comment_id")
if not comment_id:
@@ -212,9 +212,9 @@ class KuaishouMongoStoreImplement(AbstractStore):
async def store_creator(self, creator_item: Dict):
"""
存储创作者信息到MongoDB
Store creator information to MongoDB
Args:
creator_item: 创作者数据
creator_item: Creator data
"""
user_id = creator_item.get("user_id")
if not user_id:
@@ -229,7 +229,7 @@ class KuaishouMongoStoreImplement(AbstractStore):
class KuaishouExcelStoreImplement:
"""快手Excel存储实现 - 全局单例"""
"""Kuaishou Excel storage implementation - Global singleton"""
def __new__(cls, *args, **kwargs):
from store.excel_store_base import ExcelStoreBase
+10 -10
View File
@@ -21,7 +21,7 @@
# -*- coding: utf-8 -*-
# @Author : persist1@126.com
# @Time : 2025/9/5 19:34
# @Desc : 贴吧存储实现类
# @Desc : Tieba storage implementation class
import asyncio
import csv
import json
@@ -44,7 +44,7 @@ from database.mongodb_store_base import MongoDBStoreBase
def calculate_number_of_files(file_store_path: str) -> int:
"""计算数据保存文件的前部分排序数字,支持每次运行代码不写到同一个文件中
"""Calculate the prefix sorting number for data save files, supporting writing to different files for each run
Args:
file_store_path;
Returns:
@@ -203,16 +203,16 @@ class TieBaSqliteStoreImplement(TieBaDbStoreImplement):
class TieBaMongoStoreImplement(AbstractStore):
"""贴吧MongoDB存储实现"""
"""Tieba MongoDB storage implementation"""
def __init__(self):
self.mongo_store = MongoDBStoreBase(collection_prefix="tieba")
async def store_content(self, content_item: Dict):
"""
存储帖子内容到MongoDB
Store post content to MongoDB
Args:
content_item: 帖子内容数据
content_item: Post content data
"""
note_id = content_item.get("note_id")
if not note_id:
@@ -227,9 +227,9 @@ class TieBaMongoStoreImplement(AbstractStore):
async def store_comment(self, comment_item: Dict):
"""
存储评论到MongoDB
Store comment to MongoDB
Args:
comment_item: 评论数据
comment_item: Comment data
"""
comment_id = comment_item.get("comment_id")
if not comment_id:
@@ -244,9 +244,9 @@ class TieBaMongoStoreImplement(AbstractStore):
async def store_creator(self, creator_item: Dict):
"""
存储创作者信息到MongoDB
Store creator information to MongoDB
Args:
creator_item: 创作者数据
creator_item: Creator data
"""
user_id = creator_item.get("user_id")
if not user_id:
@@ -261,7 +261,7 @@ class TieBaMongoStoreImplement(AbstractStore):
class TieBaExcelStoreImplement:
"""贴吧Excel存储实现 - 全局单例"""
"""Tieba Excel storage implementation - Global singleton"""
def __new__(cls, *args, **kwargs):
from store.excel_store_base import ExcelStoreBase
+1 -1
View File
@@ -188,7 +188,7 @@ async def save_creator(user_id: str, user_info: Dict):
local_db_item = {
'user_id': user_id,
'nickname': user_info.get('screen_name'),
'gender': '女' if user_info.get('gender') == "f" else '男',
'gender': 'Female' if user_info.get('gender') == "f" else 'Male',
'avatar': user_info.get('avatar_hd'),
'desc': user_info.get('description'),
'ip_location': user_info.get("source", "").replace("来自", ""),
+10 -10
View File
@@ -21,7 +21,7 @@
# -*- coding: utf-8 -*-
# @Author : persist1@126.com
# @Time : 2025/9/5 19:34
# @Desc : 微博存储实现类
# @Desc : Weibo storage implementation class
import asyncio
import csv
import json
@@ -44,7 +44,7 @@ from database.mongodb_store_base import MongoDBStoreBase
def calculate_number_of_files(file_store_path: str) -> int:
"""计算数据保存文件的前部分排序数字,支持每次运行代码不写到同一个文件中
"""Calculate the prefix sorting number for data save files, supporting writing to different files for each run
Args:
file_store_path;
Returns:
@@ -225,16 +225,16 @@ class WeiboSqliteStoreImplement(WeiboDbStoreImplement):
class WeiboMongoStoreImplement(AbstractStore):
"""微博MongoDB存储实现"""
"""Weibo MongoDB storage implementation"""
def __init__(self):
self.mongo_store = MongoDBStoreBase(collection_prefix="weibo")
async def store_content(self, content_item: Dict):
"""
存储微博内容到MongoDB
Store Weibo content to MongoDB
Args:
content_item: 微博内容数据
content_item: Weibo content data
"""
note_id = content_item.get("note_id")
if not note_id:
@@ -249,9 +249,9 @@ class WeiboMongoStoreImplement(AbstractStore):
async def store_comment(self, comment_item: Dict):
"""
存储评论到MongoDB
Store comment to MongoDB
Args:
comment_item: 评论数据
comment_item: Comment data
"""
comment_id = comment_item.get("comment_id")
if not comment_id:
@@ -266,9 +266,9 @@ class WeiboMongoStoreImplement(AbstractStore):
async def store_creator(self, creator_item: Dict):
"""
存储创作者信息到MongoDB
Store creator information to MongoDB
Args:
creator_item: 创作者数据
creator_item: Creator data
"""
user_id = creator_item.get("user_id")
if not user_id:
@@ -283,7 +283,7 @@ class WeiboMongoStoreImplement(AbstractStore):
class WeiboExcelStoreImplement:
"""微博Excel存储实现 - 全局单例"""
"""Weibo Excel storage implementation - Global singleton"""
def __new__(cls, *args, **kwargs):
from store.excel_store_base import ExcelStoreBase
+1 -1
View File
@@ -20,7 +20,7 @@
# -*- coding: utf-8 -*-
# @Author : Erm
# @Time : 2024/4/9 17:35
# @Desc : 微博媒体保存
# @Desc : Weibo media storage
import pathlib
from typing import Dict
+53 -53
View File
@@ -50,7 +50,7 @@ class XhsStoreFactory:
def get_video_url_arr(note_item: Dict) -> List:
"""
获取视频url数组
Get video url array
Args:
note_item:
@@ -64,7 +64,7 @@ def get_video_url_arr(note_item: Dict) -> List:
originVideoKey = note_item.get('video').get('consumer').get('origin_video_key')
if originVideoKey == '':
originVideoKey = note_item.get('video').get('consumer').get('originVideoKey')
# 降级有水印
# Fallback with watermark
if originVideoKey == '':
videos = note_item.get('video').get('media').get('stream').get('h264')
if type(videos).__name__ == 'list':
@@ -77,7 +77,7 @@ def get_video_url_arr(note_item: Dict) -> List:
async def update_xhs_note(note_item: Dict):
"""
更新小红书笔记
Update Xiaohongshu note
Args:
note_item:
@@ -97,26 +97,26 @@ async def update_xhs_note(note_item: Dict):
video_url = ','.join(get_video_url_arr(note_item))
local_db_item = {
"note_id": note_item.get("note_id"), # 帖子id
"type": note_item.get("type"), # 帖子类型
"title": note_item.get("title") or note_item.get("desc", "")[:255], # 帖子标题
"desc": note_item.get("desc", ""), # 帖子描述
"video_url": video_url, # 帖子视频url
"time": note_item.get("time"), # 帖子发布时间
"last_update_time": note_item.get("last_update_time", 0), # 帖子最后更新时间
"user_id": user_info.get("user_id"), # 用户id
"nickname": user_info.get("nickname"), # 用户昵称
"avatar": user_info.get("avatar"), # 用户头像
"liked_count": interact_info.get("liked_count"), # 点赞数
"collected_count": interact_info.get("collected_count"), # 收藏数
"comment_count": interact_info.get("comment_count"), # 评论数
"share_count": interact_info.get("share_count"), # 分享数
"ip_location": note_item.get("ip_location", ""), # ip地址
"image_list": ','.join([img.get('url', '') for img in image_list]), # 图片url
"tag_list": ','.join([tag.get('name', '') for tag in tag_list if tag.get('type') == 'topic']), # 标签
"last_modify_ts": utils.get_current_timestamp(), # 最后更新时间戳(MediaCrawler程序生成的,主要用途在db存储的时候记录一条记录最新更新时间)
"note_url": f"https://www.xiaohongshu.com/explore/{note_id}?xsec_token={note_item.get('xsec_token')}&xsec_source=pc_search", # 帖子url
"source_keyword": source_keyword_var.get(), # 搜索关键词
"note_id": note_item.get("note_id"), # Note ID
"type": note_item.get("type"), # Note type
"title": note_item.get("title") or note_item.get("desc", "")[:255], # Note title
"desc": note_item.get("desc", ""), # Note description
"video_url": video_url, # Note video url
"time": note_item.get("time"), # Note publish time
"last_update_time": note_item.get("last_update_time", 0), # Note last update time
"user_id": user_info.get("user_id"), # User ID
"nickname": user_info.get("nickname"), # User nickname
"avatar": user_info.get("avatar"), # User avatar
"liked_count": interact_info.get("liked_count"), # Like count
"collected_count": interact_info.get("collected_count"), # Collection count
"comment_count": interact_info.get("comment_count"), # Comment count
"share_count": interact_info.get("share_count"), # Share count
"ip_location": note_item.get("ip_location", ""), # IP location
"image_list": ','.join([img.get('url', '') for img in image_list]), # Image URLs
"tag_list": ','.join([tag.get('name', '') for tag in tag_list if tag.get('type') == 'topic']), # Tags
"last_modify_ts": utils.get_current_timestamp(), # Last modification timestamp (Generated by MediaCrawler, mainly used to record the latest update time of a record in DB storage)
"note_url": f"https://www.xiaohongshu.com/explore/{note_id}?xsec_token={note_item.get('xsec_token')}&xsec_source=pc_search", # Note URL
"source_keyword": source_keyword_var.get(), # Search keyword
"xsec_token": note_item.get("xsec_token"), # xsec_token
}
utils.logger.info(f"[store.xhs.update_xhs_note] xhs note: {local_db_item}")
@@ -125,7 +125,7 @@ async def update_xhs_note(note_item: Dict):
async def batch_update_xhs_note_comments(note_id: str, comments: List[Dict]):
"""
批量更新小红书笔记评论
Batch update Xiaohongshu note comments
Args:
note_id:
comments:
@@ -141,7 +141,7 @@ async def batch_update_xhs_note_comments(note_id: str, comments: List[Dict]):
async def update_xhs_note_comment(note_id: str, comment_item: Dict):
"""
更新小红书笔记评论
Update Xiaohongshu note comment
Args:
note_id:
comment_item:
@@ -154,18 +154,18 @@ async def update_xhs_note_comment(note_id: str, comment_item: Dict):
comment_pictures = [item.get("url_default", "") for item in comment_item.get("pictures", [])]
target_comment = comment_item.get("target_comment", {})
local_db_item = {
"comment_id": comment_id, # 评论id
"create_time": comment_item.get("create_time"), # 评论时间
"ip_location": comment_item.get("ip_location"), # ip地址
"note_id": note_id, # 帖子id
"content": comment_item.get("content"), # 评论内容
"user_id": user_info.get("user_id"), # 用户id
"nickname": user_info.get("nickname"), # 用户昵称
"avatar": user_info.get("image"), # 用户头像
"sub_comment_count": comment_item.get("sub_comment_count", 0), # 子评论数
"pictures": ",".join(comment_pictures), # 评论图片
"parent_comment_id": target_comment.get("id", 0), # 父评论id
"last_modify_ts": utils.get_current_timestamp(), # 最后更新时间戳(MediaCrawler程序生成的,主要用途在db存储的时候记录一条记录最新更新时间)
"comment_id": comment_id, # Comment ID
"create_time": comment_item.get("create_time"), # Comment time
"ip_location": comment_item.get("ip_location"), # IP location
"note_id": note_id, # Note ID
"content": comment_item.get("content"), # Comment content
"user_id": user_info.get("user_id"), # User ID
"nickname": user_info.get("nickname"), # User nickname
"avatar": user_info.get("image"), # User avatar
"sub_comment_count": comment_item.get("sub_comment_count", 0), # Sub-comment count
"pictures": ",".join(comment_pictures), # Comment pictures
"parent_comment_id": target_comment.get("id", 0), # Parent comment ID
"last_modify_ts": utils.get_current_timestamp(), # Last modification timestamp (Generated by MediaCrawler, mainly used to record the latest update time of a record in DB storage)
"like_count": comment_item.get("like_count", 0),
}
utils.logger.info(f"[store.xhs.update_xhs_note_comment] xhs note comment:{local_db_item}")
@@ -174,7 +174,7 @@ async def update_xhs_note_comment(note_id: str, comment_item: Dict):
async def save_creator(user_id: str, creator: Dict):
"""
保存小红书创作者
Save Xiaohongshu creator
Args:
user_id:
creator:
@@ -197,25 +197,25 @@ async def save_creator(user_id: str, creator: Dict):
def get_gender(gender):
if gender == 1:
return '女'
return 'Female'
elif gender == 0:
return '男'
return 'Male'
else:
return None
local_db_item = {
'user_id': user_id, # 用户id
'nickname': user_info.get('nickname'), # 昵称
'gender': get_gender(user_info.get('gender')), # 性别
'avatar': user_info.get('images'), # 头像
'desc': user_info.get('desc'), # 个人描述
'ip_location': user_info.get('ipLocation'), # ip地址
'follows': follows, # 关注数
'fans': fans, # 粉丝数
'interaction': interaction, # 互动数
'user_id': user_id, # User ID
'nickname': user_info.get('nickname'), # Nickname
'gender': get_gender(user_info.get('gender')), # Gender
'avatar': user_info.get('images'), # Avatar
'desc': user_info.get('desc'), # Personal description
'ip_location': user_info.get('ipLocation'), # IP location
'follows': follows, # Following count
'fans': fans, # Fans count
'interaction': interaction, # Interaction count
'tag_list': json.dumps({tag.get('tagType'): tag.get('name')
for tag in creator.get('tags')}, ensure_ascii=False), # 标签
"last_modify_ts": utils.get_current_timestamp(), # 最后更新时间戳(MediaCrawler程序生成的,主要用途在db存储的时候记录一条记录最新更新时间)
for tag in creator.get('tags')}, ensure_ascii=False), # Tags
"last_modify_ts": utils.get_current_timestamp(), # Last modification timestamp (Generated by MediaCrawler, mainly used to record the latest update time of a record in DB storage)
}
utils.logger.info(f"[store.xhs.save_creator] creator:{local_db_item}")
await XhsStoreFactory.create_store().store_creator(local_db_item)
@@ -223,7 +223,7 @@ async def save_creator(user_id: str, creator: Dict):
async def update_xhs_note_image(note_id, pic_content, extension_file_name):
"""
更新小红书笔记图片
Update Xiaohongshu note image
Args:
note_id:
pic_content:
@@ -238,7 +238,7 @@ async def update_xhs_note_image(note_id, pic_content, extension_file_name):
async def update_xhs_note_video(note_id, video_content, extension_file_name):
"""
更新小红书笔记视频
Update Xiaohongshu note video
Args:
note_id:
video_content:
+9 -9
View File
@@ -18,7 +18,7 @@
# @Author : persist1@126.com
# @Time : 2025/9/5 19:34
# @Desc : 小红书存储实现类
# @Desc : Xiaohongshu storage implementation class
import json
import os
from datetime import datetime
@@ -281,7 +281,7 @@ class XhsSqliteStoreImplement(XhsDbStoreImplement):
class XhsMongoStoreImplement(AbstractStore):
"""小红书MongoDB存储实现"""
"""Xiaohongshu MongoDB storage implementation"""
def __init__(self, **kwargs):
super().__init__(**kwargs)
@@ -289,9 +289,9 @@ class XhsMongoStoreImplement(AbstractStore):
async def store_content(self, content_item: Dict):
"""
存储笔记内容到MongoDB
Store note content to MongoDB
Args:
content_item: 笔记内容数据
content_item: Note content data
"""
note_id = content_item.get("note_id")
if not note_id:
@@ -306,9 +306,9 @@ class XhsMongoStoreImplement(AbstractStore):
async def store_comment(self, comment_item: Dict):
"""
存储评论到MongoDB
Store comment to MongoDB
Args:
comment_item: 评论数据
comment_item: Comment data
"""
comment_id = comment_item.get("comment_id")
if not comment_id:
@@ -323,9 +323,9 @@ class XhsMongoStoreImplement(AbstractStore):
async def store_creator(self, creator_item: Dict):
"""
存储创作者信息到MongoDB
Store creator information to MongoDB
Args:
creator_item: 创作者数据
creator_item: Creator data
"""
user_id = creator_item.get("user_id")
if not user_id:
@@ -340,7 +340,7 @@ class XhsMongoStoreImplement(AbstractStore):
class XhsExcelStoreImplement:
"""小红书Excel存储实现 - 全局单例"""
"""Xiaohongshu Excel storage implementation - Global singleton"""
def __new__(cls, *args, **kwargs):
from store.excel_store_base import ExcelStoreBase
+1 -1
View File
@@ -20,7 +20,7 @@
# -*- coding: utf-8 -*-
# @Author : helloteemo
# @Time : 2024/7/11 22:35
# @Desc : 小红书媒体保存
# @Desc : Xiaohongshu media storage
import pathlib
from typing import Dict
+5 -5
View File
@@ -53,7 +53,7 @@ class ZhihuStoreFactory:
async def batch_update_zhihu_contents(contents: List[ZhihuContent]):
"""
批量更新知乎内容
Batch update Zhihu contents
Args:
contents:
@@ -68,7 +68,7 @@ async def batch_update_zhihu_contents(contents: List[ZhihuContent]):
async def update_zhihu_content(content_item: ZhihuContent):
"""
更新知乎内容
Update Zhihu content
Args:
content_item:
@@ -85,7 +85,7 @@ async def update_zhihu_content(content_item: ZhihuContent):
async def batch_update_zhihu_note_comments(comments: List[ZhihuComment]):
"""
批量更新知乎内容评论
Batch update Zhihu content comments
Args:
comments:
@@ -101,7 +101,7 @@ async def batch_update_zhihu_note_comments(comments: List[ZhihuComment]):
async def update_zhihu_content_comment(comment_item: ZhihuComment):
"""
更新知乎内容评论
Update Zhihu content comment
Args:
comment_item:
@@ -116,7 +116,7 @@ async def update_zhihu_content_comment(comment_item: ZhihuComment):
async def save_creator(creator: ZhihuCreator):
"""
保存知乎创作者信息
Save Zhihu creator information
Args:
creator:
+10 -10
View File
@@ -21,7 +21,7 @@
# -*- coding: utf-8 -*-
# @Author : persist1@126.com
# @Time : 2025/9/5 19:34
# @Desc : 知乎存储实现类
# @Desc : Zhihu storage implementation class
import asyncio
import csv
import json
@@ -43,7 +43,7 @@ from tools.async_file_writer import AsyncFileWriter
from database.mongodb_store_base import MongoDBStoreBase
def calculate_number_of_files(file_store_path: str) -> int:
"""计算数据保存文件的前部分排序数字,支持每次运行代码不写到同一个文件中
"""Calculate the prefix sorting number for data save files, supporting writing to different files for each run
Args:
file_store_path;
Returns:
@@ -202,16 +202,16 @@ class ZhihuSqliteStoreImplement(ZhihuDbStoreImplement):
class ZhihuMongoStoreImplement(AbstractStore):
"""知乎MongoDB存储实现"""
"""Zhihu MongoDB storage implementation"""
def __init__(self):
self.mongo_store = MongoDBStoreBase(collection_prefix="zhihu")
async def store_content(self, content_item: Dict):
"""
存储内容到MongoDB
Store content to MongoDB
Args:
content_item: 内容数据
content_item: Content data
"""
note_id = content_item.get("note_id")
if not note_id:
@@ -226,9 +226,9 @@ class ZhihuMongoStoreImplement(AbstractStore):
async def store_comment(self, comment_item: Dict):
"""
存储评论到MongoDB
Store comment to MongoDB
Args:
comment_item: 评论数据
comment_item: Comment data
"""
comment_id = comment_item.get("comment_id")
if not comment_id:
@@ -243,9 +243,9 @@ class ZhihuMongoStoreImplement(AbstractStore):
async def store_creator(self, creator_item: Dict):
"""
存储创作者信息到MongoDB
Store creator information to MongoDB
Args:
creator_item: 创作者数据
creator_item: Creator data
"""
user_id = creator_item.get("user_id")
if not user_id:
@@ -260,7 +260,7 @@ class ZhihuMongoStoreImplement(AbstractStore):
class ZhihuExcelStoreImplement:
"""知乎Excel存储实现 - 全局单例"""
"""Zhihu Excel storage implementation - Global singleton"""
def __new__(cls, *args, **kwargs):
from store.excel_store_base import ExcelStoreBase