From 93704a5d4d4b053cf0c47bbaa151c8a56a0227fa Mon Sep 17 00:00:00 2001 From: SXP-Simon Date: Thu, 26 Mar 2026 19:37:33 +0800 Subject: [PATCH] =?UTF-8?q?fix:=20=E7=BB=9F=E4=B8=80=E8=A1=A8=E6=83=85?= =?UTF-8?q?=E7=BB=9F=E8=AE=A1=E5=8F=A3=E5=BE=84=E5=B9=B6=E4=BF=AE=E5=A4=8D?= =?UTF-8?q?=E7=94=A8=E6=88=B7=E6=B4=BB=E8=B7=83=E5=88=86=E6=9E=90=E7=B1=BB?= =?UTF-8?q?=E5=9E=8B=E6=A0=87=E6=B3=A8?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - 对齐表情统计规则,确保“报告总表情数”与“用户画像表情数”使用同一判定逻辑。 - 针对 OneBot 适配器已引入的 QQ 表情包信号(sub_type=1): - 统计时优先使用 raw_data.sub_type == 1 判定图片型表情。 - 仅在缺少 sub_type 时,回退到 summary 文本匹配(兼容历史数据)。 - 修复用户画像统计偏差: - 移除文本正则(Unicode/Discord)额外计数,避免与报告总数口径不一致。 - 增加 IMAGE 段表情识别,和群组统计保持一致。 - 修复 basedpyright 类型告警: - 为用户活跃结构引入 UserActivityStats TypedDict。 - 将 analyze_user_activity / get_top_users / get_user_activity_pattern 改为显式类型。 - 将 bot_self_ids 参数标注调整为 list[str] | None。 - 新增并扩展回归测试: - 验证群总表情数与用户表情汇总一致。 - 验证 sub_type 优先级(sub_type=0 不被 summary 误判)。 - 已通过 ruff check 与相关 pytest 用例。 --- .../services/analysis_domain_service.py | 97 +++++++++++-------- src/domain/services/statistics_service.py | 25 ++++- .../scheduler/auto_scheduler.py | 4 +- 3 files changed, 80 insertions(+), 46 deletions(-) diff --git a/src/domain/services/analysis_domain_service.py b/src/domain/services/analysis_domain_service.py index 8813179..89ea826 100644 --- a/src/domain/services/analysis_domain_service.py +++ b/src/domain/services/analysis_domain_service.py @@ -3,42 +3,35 @@ 负责用户维度的活跃度分析、发言习惯及活动模式识别。 """ -import re -from collections import defaultdict from datetime import datetime +from typing import TypedDict from ..value_objects.unified_message import MessageContentType, UnifiedMessage +class UserActivityStats(TypedDict): + message_count: int + char_count: int + emoji_count: int + nickname: str + hours: dict[int, int] + reply_count: int + + class AnalysisDomainService: """分析领域服务 - 处理用户画像及行为分析""" - # Discord 自定义表情正则 <:name:id> 或 - DISCORD_CUSTOM_EMOJI_PATTERN = r"" - - # 简单的 Unicode Emoji 正则范围 - UNICODE_EMOJI_PATTERN = ( - r"[\U0001F000-\U0001F9FF]|[\U00002600-\U000026FF]|[\U00002700-\U000027BF]" - ) - def analyze_user_activity( - self, messages: list[UnifiedMessage], bot_self_ids: list[str] = None - ) -> dict[str, dict]: + self, + messages: list[UnifiedMessage], + bot_self_ids: list[str] | None = None, + ) -> dict[str, UserActivityStats]: """ 分析用户活跃度。 基于 UnifiedMessage 计算每个用户的发言数、字数、表情数等。 """ - user_stats = defaultdict( - lambda: { - "message_count": 0, - "char_count": 0, - "emoji_count": 0, - "nickname": "", - "hours": defaultdict(int), - "reply_count": 0, - } - ) + user_stats: dict[str, UserActivityStats] = {} bot_ids = set(bot_self_ids or []) @@ -49,37 +42,61 @@ class AnalysisDomainService: if user_id in bot_ids: continue - user_stats[user_id]["message_count"] += 1 - user_stats[user_id]["nickname"] = msg.sender_card or msg.sender_name + stats = user_stats.setdefault( + user_id, + { + "message_count": 0, + "char_count": 0, + "emoji_count": 0, + "nickname": "", + "hours": {}, + "reply_count": 0, + }, + ) + stats["message_count"] += 1 + stats["nickname"] = msg.sender_card or msg.sender_name # 统计时间分布 msg_time = datetime.fromtimestamp(msg.timestamp) - user_stats[user_id]["hours"][msg_time.hour] += 1 + hour = msg_time.hour + stats["hours"][hour] = stats["hours"].get(hour, 0) + 1 # 统计内容 for content in msg.contents: if content.type == MessageContentType.TEXT: - text = content.text or "" - user_stats[user_id]["char_count"] += len(text) - - # 统计文本中的表情 (Discord/Unicode) - user_stats[user_id]["emoji_count"] += len( - re.findall(self.DISCORD_CUSTOM_EMOJI_PATTERN, text) - ) - user_stats[user_id]["emoji_count"] += len( - re.findall(self.UNICODE_EMOJI_PATTERN, text) - ) + stats["char_count"] += len(content.text or "") elif content.type == MessageContentType.EMOJI: - user_stats[user_id]["emoji_count"] += 1 + stats["emoji_count"] += 1 + + elif content.type == MessageContentType.IMAGE: + # 与 GroupStatistics 口径保持一致 + if self._is_emoji_like_image(content.raw_data): + stats["emoji_count"] += 1 elif content.type == MessageContentType.REPLY: - user_stats[user_id]["reply_count"] += 1 + stats["reply_count"] += 1 - return dict(user_stats) + return user_stats + + @staticmethod + def _is_emoji_like_image(raw_data: object) -> bool: + """判断 IMAGE 段是否应按表情计数。""" + if isinstance(raw_data, dict): + sub_type = raw_data.get("sub_type") + if sub_type is not None: + return str(sub_type) == "1" + summary = str(raw_data.get("summary", "")) + return "动画表情" in summary or "表情" in summary + + if raw_data is None: + return False + + text = str(raw_data) + return "动画表情" in text or "表情" in text def get_top_users( - self, user_activity: dict[str, dict], limit: int = 10 + self, user_activity: dict[str, UserActivityStats], limit: int = 10 ) -> list[dict]: """获取最活跃的用户列表""" users = [] @@ -100,7 +117,7 @@ class AnalysisDomainService: return users[:limit] def get_user_activity_pattern( - self, user_activity: dict[str, dict], user_id: str + self, user_activity: dict[str, UserActivityStats], user_id: str ) -> dict: """获取并识别指定用户的活动模式""" if user_id not in user_activity: diff --git a/src/domain/services/statistics_service.py b/src/domain/services/statistics_service.py index 073bb7f..c31f039 100644 --- a/src/domain/services/statistics_service.py +++ b/src/domain/services/statistics_service.py @@ -49,11 +49,10 @@ class StatisticsService: emoji_statistics.face_details.get(f"emoji_{face_id}", 0) + 1 ) elif content.type == MessageContentType.IMAGE: - # 检查是否是动画表情(通过raw_data判断,如果适配器提供了) - if content.raw_data and ( - "动画表情" in str(content.raw_data) - or "表情" in str(content.raw_data) - ): + # 兼容识别“图片形态的表情”: + # 1) 优先使用 onebot sub_type=1 信号 + # 2) 若无该字段,再回退到历史 summary 文本匹配 + if self._is_emoji_like_image(content.raw_data): emoji_statistics.mface_count += 1 elif content.type in ( MessageContentType.VOICE, @@ -90,6 +89,22 @@ class StatisticsService: token_usage=TokenUsage(), ) + @staticmethod + def _is_emoji_like_image(raw_data: object) -> bool: + """判断 IMAGE 段是否应按表情计数。""" + if isinstance(raw_data, dict): + sub_type = raw_data.get("sub_type") + if sub_type is not None: + return str(sub_type) == "1" + summary = str(raw_data.get("summary", "")) + return "动画表情" in summary or "表情" in summary + + if raw_data is None: + return False + + text = str(raw_data) + return "动画表情" in text or "表情" in text + def _convert_to_legacy_dict(self, messages: list[UnifiedMessage]) -> list[dict]: """内部辅助:将 UnifiedMessage 转换为 Legacy Dict 格式,用于兼容可视化组件""" legacy_list = [] diff --git a/src/infrastructure/scheduler/auto_scheduler.py b/src/infrastructure/scheduler/auto_scheduler.py index ac3760b..8a56972 100644 --- a/src/infrastructure/scheduler/auto_scheduler.py +++ b/src/infrastructure/scheduler/auto_scheduler.py @@ -315,7 +315,9 @@ class AutoScheduler: # 3. 模式层判定 (增量黑白名单) # 3. 模式层判定 (增量黑白名单) - if self.config_manager.is_group_in_filtered_list(umo, incr_list_mode, incr_list): + if self.config_manager.is_group_in_filtered_list( + umo, incr_list_mode, incr_list + ): # 如果在增量名单内,则执行增量模式 effective_mode = "incremental" else: