refactor: 教学版移除全平台用户个人信息采集与持久化

- 用户 ID 转为匿名 creator_hash,昵称中间脱敏,IP/头像/主页/签名/性别不再采集
- 覆盖 xhs/weibo/bilibili/douyin/kuaishou/tieba/zhihu 7 个平台
- 删除 7 张 creator 档案 ORM 表,15 张内容/评论表新增 creator_hash 列
- B 站禁用粉丝/关注/联系人列表抓取
- 新增 tools/user_hash.py 与 4 个平台的 mock+SQLite 端到端测试

测试: pytest tests/test_no_user_info.py tests/test_weibo_no_user_info.py tests/test_douyin_no_user_info.py tests/test_kuaishou_no_user_info.py (21 passed)
This commit is contained in:
程序员阿江(Relakkes)
2026-07-01 13:09:55 +08:00
parent 8b4d8fcffa
commit 9f4f8bf768
26 changed files with 1426 additions and 822 deletions
+10 -10
View File
@@ -459,11 +459,11 @@ class ZhiHuClient(AbstractApiClient, ProxyRefreshMixin):
}
return await self.get(uri, params)
async def get_all_anwser_by_creator(self, creator: ZhihuCreator, crawl_interval: float = 1.0, callback: Optional[Callable] = None) -> List[ZhihuContent]:
async def get_all_anwser_by_creator(self, url_token: str, crawl_interval: float = 1.0, callback: Optional[Callable] = None) -> List[ZhihuContent]:
"""
Get all answers by creator
Args:
creator: Creator information
url_token: Creator url token (in-memory only, not persisted)
crawl_interval: Crawl delay interval in seconds
callback: Callback after completing one crawl
@@ -475,10 +475,10 @@ class ZhiHuClient(AbstractApiClient, ProxyRefreshMixin):
offset: int = 0
limit: int = 20
while not is_end:
res = await self.get_creator_answers(creator.url_token, offset, limit)
res = await self.get_creator_answers(url_token, offset, limit)
if not res:
break
utils.logger.info(f"[ZhiHuClient.get_all_anwser_by_creator] Get creator {creator.url_token} answers: {res}")
utils.logger.info(f"[ZhiHuClient.get_all_anwser_by_creator] Get creator {url_token} answers: {res}")
paging_info = res.get("paging", {})
is_end = paging_info.get("is_end")
contents = self._extractor.extract_content_list_from_creator(res.get("data"))
@@ -491,14 +491,14 @@ class ZhiHuClient(AbstractApiClient, ProxyRefreshMixin):
async def get_all_articles_by_creator(
self,
creator: ZhihuCreator,
url_token: str,
crawl_interval: float = 1.0,
callback: Optional[Callable] = None,
) -> List[ZhihuContent]:
"""
Get all articles by creator
Args:
creator:
url_token: Creator url token (in-memory only, not persisted)
crawl_interval:
callback:
@@ -510,7 +510,7 @@ class ZhiHuClient(AbstractApiClient, ProxyRefreshMixin):
offset: int = 0
limit: int = 20
while not is_end:
res = await self.get_creator_articles(creator.url_token, offset, limit)
res = await self.get_creator_articles(url_token, offset, limit)
if not res:
break
paging_info = res.get("paging", {})
@@ -525,14 +525,14 @@ class ZhiHuClient(AbstractApiClient, ProxyRefreshMixin):
async def get_all_videos_by_creator(
self,
creator: ZhihuCreator,
url_token: str,
crawl_interval: float = 1.0,
callback: Optional[Callable] = None,
) -> List[ZhihuContent]:
"""
Get all videos by creator
Args:
creator:
url_token: Creator url token (in-memory only, not persisted)
crawl_interval:
callback:
@@ -544,7 +544,7 @@ class ZhiHuClient(AbstractApiClient, ProxyRefreshMixin):
offset: int = 0
limit: int = 20
while not is_end:
res = await self.get_creator_videos(creator.url_token, offset, limit)
res = await self.get_creator_videos(url_token, offset, limit)
if not res:
break
paging_info = res.get("paging", {})
+3 -4
View File
@@ -276,27 +276,26 @@ class ZhihuCrawler(AbstractCrawler):
utils.logger.info(
f"[ZhihuCrawler.get_creators_and_notes] Creator info: {createor_info}"
)
await zhihu_store.save_creator(creator=createor_info)
# By default, only answer information is extracted, uncomment below if articles and videos are needed
# Get all anwser information of the creator
all_content_list = await self.zhihu_client.get_all_anwser_by_creator(
creator=createor_info,
url_token=user_url_token,
crawl_interval=config.CRAWLER_MAX_SLEEP_SEC,
callback=zhihu_store.batch_update_zhihu_contents,
)
# Get all articles of the creator's contents
# all_content_list = await self.zhihu_client.get_all_articles_by_creator(
# creator=createor_info,
# url_token=user_url_token,
# crawl_interval=config.CRAWLER_MAX_SLEEP_SEC,
# callback=zhihu_store.batch_update_zhihu_contents
# )
# Get all videos of the creator's contents
# all_content_list = await self.zhihu_client.get_all_videos_by_creator(
# creator=createor_info,
# url_token=user_url_token,
# crawl_interval=config.CRAWLER_MAX_SLEEP_SEC,
# callback=zhihu_store.batch_update_zhihu_contents
# )
+9 -28
View File
@@ -30,6 +30,7 @@ from constant import zhihu as zhihu_constant
from model.m_zhihu import ZhihuComment, ZhihuContent, ZhihuCreator
from tools import utils
from tools.crawler_util import extract_text_from_html
from tools.user_hash import anonymize_user_id, mask_nickname
ZHIHU_SGIN_JS = None
@@ -120,11 +121,8 @@ class ZhihuExtractor:
# extract author info
author_info = self._extract_content_or_comment_author(answer.get("author"))
res.user_id = author_info.user_id
res.user_link = author_info.user_link
res.creator_hash = author_info.creator_hash
res.user_nickname = author_info.user_nickname
res.user_avatar = author_info.user_avatar
res.user_url_token = author_info.url_token
return res
def _extract_article_content(self, article: Dict) -> ZhihuContent:
@@ -150,11 +148,8 @@ class ZhihuExtractor:
# extract author info
author_info = self._extract_content_or_comment_author(article.get("author"))
res.user_id = author_info.user_id
res.user_link = author_info.user_link
res.creator_hash = author_info.creator_hash
res.user_nickname = author_info.user_nickname
res.user_avatar = author_info.user_avatar
res.user_url_token = author_info.url_token
return res
def _extract_zvideo_content(self, zvideo: Dict) -> ZhihuContent:
@@ -184,11 +179,8 @@ class ZhihuExtractor:
# extract author info
author_info = self._extract_content_or_comment_author(zvideo.get("author"))
res.user_id = author_info.user_id
res.user_link = author_info.user_link
res.creator_hash = author_info.creator_hash
res.user_nickname = author_info.user_nickname
res.user_avatar = author_info.user_avatar
res.user_url_token = author_info.url_token
return res
@staticmethod
@@ -207,11 +199,8 @@ class ZhihuExtractor:
return res
if not author.get("id"):
author = author.get("member")
res.user_id = author.get("id")
res.user_link = f"{zhihu_constant.ZHIHU_URL}/people/{author.get('url_token')}"
res.user_nickname = author.get("name")
res.user_avatar = author.get("avatar_url")
res.url_token = author.get("url_token")
res.creator_hash = anonymize_user_id(author.get("id"))
res.user_nickname = mask_nickname(author.get("name"))
except Exception as e :
utils.logger.warning(
@@ -253,7 +242,6 @@ class ZhihuExtractor:
res.parent_comment_id = comment.get("reply_comment_id")
res.content = extract_text_from_html(comment.get("content"))
res.publish_time = comment.get("created_time")
res.ip_location = self._extract_comment_ip_location(comment.get("comment_tag", []))
res.sub_comment_count = comment.get("child_comment_count")
res.like_count = comment.get("like_count") if comment.get("like_count") else 0
res.dislike_count = comment.get("dislike_count") if comment.get("dislike_count") else 0
@@ -262,10 +250,8 @@ class ZhihuExtractor:
# extract author info
author_info = self._extract_content_or_comment_author(comment.get("author"))
res.user_id = author_info.user_id
res.user_link = author_info.user_link
res.creator_hash = author_info.creator_hash
res.user_nickname = author_info.user_nickname
res.user_avatar = author_info.user_avatar
return res
@staticmethod
@@ -352,13 +338,8 @@ class ZhihuExtractor:
return None
res = ZhihuCreator()
res.user_id = creator_info.get("id")
res.user_link = f"{zhihu_constant.ZHIHU_URL}/people/{user_url_token}"
res.user_nickname = creator_info.get("name")
res.user_avatar = creator_info.get("avatarUrl")
res.url_token = creator_info.get("urlToken") or user_url_token
res.gender = self._foramt_gender_text(creator_info.get("gender"))
res.ip_location = creator_info.get("ipInfo")
res.creator_hash = anonymize_user_id(creator_info.get("id"))
res.user_nickname = mask_nickname(creator_info.get("name"))
res.follows = creator_info.get("followingCount")
res.fans = creator_info.get("followerCount")
res.anwser_count = creator_info.get("answerCount")