refactor: 数据存储重构，分离不同类型的存储实现

2026-06-09 03:17:25 +08:00 · 2024-01-14 22:06:31 +08:00
parent e31aebbdfb
commit 894dabcf63
37 changed files with 1427 additions and 864 deletions
--- a/store/weibo/init.py
+++ b/store/weibo/init.py
@@ -0,0 +1,88 @@
+# -*- coding: utf-8 -*-
+# @Author  : relakkes@gmail.com
+# @Time    : 2024/1/14 21:34
+# @Desc    :
+
+from typing import List
+
+import config
+
+from .weibo_store_db_types import *
+from .weibo_store_impl import *
+
+
+class WeibostoreFactory:
+    STORES = {
+        "csv": WeiboCsvStoreImplement,
+        "db": WeiboDbStoreImplement,
+        "json": BiliJsonStoreImplement
+    }
+
+    @staticmethod
+    def create_store() -> AbstractStore:
+        store_class = WeibostoreFactory.STORES.get(config.SAVE_DATA_OPTION)
+        if not store_class:
+            raise ValueError(
+                "[WeibotoreFactory.create_store] Invalid save option only supported csv or db or json ...")
+        return store_class()
+
+async def update_weibo_note(note_item: Dict):
+    mblog: Dict = note_item.get("mblog")
+    user_info: Dict = mblog.get("user")
+    note_id = mblog.get("id")
+    save_content_item = {
+        # 微博信息
+        "note_id": note_id,
+        "content": mblog.get("text"),
+        "create_time": utils.rfc2822_to_timestamp(mblog.get("created_at")),
+        "create_date_time": str(utils.rfc2822_to_china_datetime(mblog.get("created_at"))),
+        "liked_count": str(mblog.get("attitudes_count", 0)),
+        "comments_count": str(mblog.get("comments_count", 0)),
+        "shared_count": str(mblog.get("reposts_count", 0)),
+        "last_modify_ts": utils.get_current_timestamp(),
+        "note_url": f"https://m.weibo.cn/detail/{note_id}",
+        "ip_location": mblog.get("region_name", "").replace("发布于 ", ""),
+
+        # 用户信息
+        "user_id": str(user_info.get("id")),
+        "nickname": user_info.get("screen_name", ""),
+        "gender": user_info.get("gender", ""),
+        "profile_url": user_info.get("profile_url", ""),
+        "avatar": user_info.get("profile_image_url", ""),
+    }
+    utils.logger.info(
+        f"[store.weibo.update_weibo_note] weibo note id:{note_id}, title:{save_content_item.get('content')[:24]} ...")
+    await WeibostoreFactory.create_store().store_content(content_item=save_content_item)
+
+
+async def batch_update_weibo_note_comments(note_id: str, comments: List[Dict]):
+    if not comments:
+        return
+    for comment_item in comments:
+        await update_weibo_note_comment(note_id, comment_item)
+
+
+async def update_weibo_note_comment(note_id: str, comment_item: Dict):
+    comment_id = str(comment_item.get("id"))
+    user_info: Dict = comment_item.get("user")
+    save_comment_item = {
+        "comment_id": comment_id,
+        "create_time": utils.rfc2822_to_timestamp(comment_item.get("created_at")),
+        "create_date_time": str(utils.rfc2822_to_china_datetime(comment_item.get("created_at"))),
+        "note_id": note_id,
+        "content": comment_item.get("text"),
+        "sub_comment_count": str(comment_item.get("total_number", 0)),
+        "comment_like_count": str(comment_item.get("like_count", 0)),
+        "last_modify_ts": utils.get_current_timestamp(),
+        "ip_location": comment_item.get("source", "").replace("来自", ""),
+
+        # 用户信息
+        "user_id": str(user_info.get("id")),
+        "nickname": user_info.get("screen_name", ""),
+        "gender": user_info.get("gender", ""),
+        "profile_url": user_info.get("profile_url", ""),
+        "avatar": user_info.get("profile_image_url", ""),
+    }
+    utils.logger.info(
+        f"[store.weibo.update_weibo_note_comment] Weibo note comment: {comment_id}, content: {save_comment_item.get('content','')[:24]} ...")
+    await WeibostoreFactory.create_store().store_comment(comment_item=save_comment_item)
--- a/store/weibo/weibo_store_db_types.py
+++ b/store/weibo/weibo_store_db_types.py
@@ -0,0 +1,57 @@
+# -*- coding: utf-8 -*-
+# @Author  : relakkes@gmail.com
+# @Time    : 2024/1/14 21:35
+# @Desc    : 微博存储到DB的模型类集合
+
+from tortoise import fields
+from tortoise.models import Model
+
+
+class WeiboBaseModel(Model):
+    id = fields.IntField(pk=True, autoincrement=True, description="自增ID")
+    user_id = fields.CharField(null=True, max_length=64, description="用户ID")
+    nickname = fields.CharField(null=True, max_length=64, description="用户昵称")
+    avatar = fields.CharField(null=True, max_length=255, description="用户头像地址")
+    gender = fields.CharField(null=True, max_length=12, description="用户性别")
+    profile_url = fields.CharField(null=True, max_length=255, description="用户主页地址")
+    ip_location = fields.CharField(null=True, max_length=32, default="发布微博的地理信息")
+    add_ts = fields.BigIntField(description="记录添加时间戳")
+    last_modify_ts = fields.BigIntField(description="记录最后修改时间戳")
+
+    class Meta:
+        abstract = True
+
+
+class WeiboNote(WeiboBaseModel):
+    note_id = fields.CharField(max_length=64, index=True, description="帖子ID")
+    content = fields.TextField(null=True, description="帖子正文内容")
+    create_time = fields.BigIntField(description="帖子发布时间戳", index=True)
+    create_date_time = fields.CharField(description="帖子发布日期时间", max_length=32, index=True)
+    liked_count = fields.CharField(null=True, max_length=16, description="帖子点赞数")
+    comments_count = fields.CharField(null=True, max_length=16, description="帖子评论数量")
+    shared_count = fields.CharField(null=True, max_length=16, description="帖子转发数量")
+    note_url = fields.CharField(null=True, max_length=512, description="帖子详情URL")
+
+    class Meta:
+        table = "weibo_note"
+        table_description = "微博帖子"
+
+    def __str__(self):
+        return f"{self.note_id}"
+
+
+class WeiboComment(WeiboBaseModel):
+    comment_id = fields.CharField(max_length=64, index=True, description="评论ID")
+    note_id = fields.CharField(max_length=64, index=True, description="帖子ID")
+    content = fields.TextField(null=True, description="评论内容")
+    create_time = fields.BigIntField(description="评论时间戳")
+    create_date_time = fields.CharField(description="评论日期时间", max_length=32, index=True)
+    comment_like_count = fields.CharField(max_length=16, description="评论点赞数量")
+    sub_comment_count = fields.CharField(max_length=16, description="评论回复数")
+
+    class Meta:
+        table = "weibo_note_comment"
+        table_description = "微博帖子评论"
+
+    def __str__(self):
+        return f"{self.comment_id}"
--- a/store/weibo/weibo_store_impl.py
+++ b/store/weibo/weibo_store_impl.py
@@ -0,0 +1,128 @@
+# -*- coding: utf-8 -*-
+# @Author  : relakkes@gmail.com
+# @Time    : 2024/1/14 21:35
+# @Desc    : 微博存储实现类
+import csv
+import pathlib
+from typing import Dict
+
+import aiofiles
+from tortoise.contrib.pydantic import pydantic_model_creator
+
+from base.base_crawler import AbstractStore
+from tools import utils
+from var import crawler_type_var
+
+
+class WeiboCsvStoreImplement(AbstractStore):
+    csv_store_path: str = "data/weibo"
+
+    def make_save_file_name(self, store_type: str) -> str:
+        """
+        make save file name by store type
+        Args:
+            store_type: contents or comments
+
+        Returns: eg: data/bilibili/search_comments_20240114.csv ...
+
+        """
+        return f"{self.csv_store_path}/{crawler_type_var.get()}_{store_type}_{utils.get_current_date()}.csv"
+
+    async def save_data_to_csv(self, save_item: Dict, store_type: str):
+        """
+        Below is a simple way to save it in CSV format.
+        Args:
+            save_item:  save content dict info
+            store_type: Save type contains content and comments（contents | comments）
+
+        Returns: no returns
+
+        """
+        pathlib.Path(self.csv_store_path).mkdir(parents=True, exist_ok=True)
+        save_file_name = self.make_save_file_name(store_type=store_type)
+        async with aiofiles.open(save_file_name, mode='a+', encoding="utf-8-sig", newline="") as f:
+            writer = csv.writer(f)
+            if await f.tell() == 0:
+                await writer.writerow(save_item.keys())
+            await writer.writerow(save_item.values())
+
+    async def store_content(self, content_item: Dict):
+        """
+        Weibo content CSV storage implementation
+        Args:
+            content_item: note item dict
+
+        Returns:
+
+        """
+        await self.save_data_to_csv(save_item=content_item, store_type="contents")
+
+    async def store_comment(self, comment_item: Dict):
+        """
+        Weibo comment CSV storage implementation
+        Args:
+            comment_item: comment item dict
+
+        Returns:
+
+        """
+        await self.save_data_to_csv(save_item=comment_item, store_type="comments")
+
+
+class WeiboDbStoreImplement(AbstractStore):
+    async def store_content(self, content_item: Dict):
+        """
+        Weibo content DB storage implementation
+        Args:
+            content_item: content item dict
+
+        Returns:
+
+        """
+        from .weibo_store_db_types import WeiboNote
+        note_id = content_item.get("note_id")
+        if not await WeiboNote.filter(note_id=note_id).exists():
+            content_item["add_ts"] = utils.get_current_timestamp()
+            weibo_note_pydantic = pydantic_model_creator(WeiboNote, name='WeiboNoteCreate', exclude=('id',))
+            weibo_data = weibo_note_pydantic(**content_item)
+            weibo_note_pydantic.model_validate(weibo_data)
+            await WeiboNote.create(**weibo_data.model_dump())
+        else:
+            weibo_note_pydantic = pydantic_model_creator(WeiboNote, name='WeiboNoteUpdate',
+                                                         exclude=('id', 'add_ts'))
+            weibo_data = weibo_note_pydantic(**content_item)
+            weibo_note_pydantic.model_validate(weibo_data)
+            await WeiboNote.filter(note_id=note_id).update(**weibo_data.model_dump())
+
+    async def store_comment(self, comment_item: Dict):
+        """
+        Weibo content DB storage implementation
+        Args:
+            comment_item: comment item dict
+
+        Returns:
+
+        """
+        from .weibo_store_db_types import WeiboComment
+        comment_id = comment_item.get("comment_id")
+        if not await WeiboComment.filter(comment_id=comment_id).exists():
+            comment_item["add_ts"] = utils.get_current_timestamp()
+            comment_pydantic = pydantic_model_creator(WeiboComment, name='WeiboNoteCommentCreate',
+                                                      exclude=('id',))
+            comment_data = comment_pydantic(**comment_item)
+            comment_pydantic.model_validate(comment_data)
+            await WeiboComment.create(**comment_data.model_dump())
+        else:
+            comment_pydantic = pydantic_model_creator(WeiboComment, name='WeiboNoteCommentUpdate',
+                                                      exclude=('id', 'add_ts'))
+            comment_data = comment_pydantic(**comment_item)
+            comment_pydantic.model_validate(comment_data)
+            await WeiboComment.filter(comment_id=comment_id).update(**comment_data.model_dump())
+
+
+class BiliJsonStoreImplement(AbstractStore):
+    async def store_content(self, content_item: Dict):
+        pass
+
+    async def store_comment(self, comment_item: Dict):
+        pass