Repository navigation
fix(analysis): 表达学习批量链路合并 Bot 回复并修正人格兼容性参数 (#257) #258
New issue
Have a question about this project? Sign up for a free GitHub account to open an issue and contact its maintainers and the community.
By clicking “Sign up for GitHub”, you agree to our terms of service and privacy statement. We’ll occasionally send you account related emails.
Already on GitHub? Sign in to your account
Changes from all commits
File filter
Filter by extension
Conversations
Jump to
Diff view
Diff view
There are no files selected for viewing
| Original file line number | Diff line number | Diff line change |
|---|---|---|
|
|
@@ -112,7 +112,110 @@ def _expression_scope_key( | |
| user_id: Optional[str] = None, | ||
| ) -> str: | ||
| return f"{group_id}:{persona_id}:{user_id or 'group-level'}" | ||
|
|
||
|
|
||
| @staticmethod | ||
| def _msg_sender_id(msg: Any) -> str: | ||
| if hasattr(msg, "sender_id"): | ||
| return str(getattr(msg, "sender_id", "") or "") | ||
| if isinstance(msg, dict): | ||
| return str(msg.get("sender_id", "") or "") | ||
| return "" | ||
|
|
||
| @staticmethod | ||
| def _msg_timestamp(msg: Any) -> float: | ||
| value = ( | ||
| getattr(msg, "timestamp", 0) | ||
| if hasattr(msg, "timestamp") | ||
| else msg.get("timestamp", 0) if isinstance(msg, dict) else 0 | ||
| ) | ||
| try: | ||
| return float(value or 0) | ||
| except (TypeError, ValueError): | ||
| return 0.0 | ||
|
|
||
| @staticmethod | ||
| def _msg_content(msg: Any) -> str: | ||
| if hasattr(msg, "message"): | ||
| return str(getattr(msg, "message", "") or "") | ||
| if isinstance(msg, dict): | ||
| return str(msg.get("message", "") or "") | ||
| return "" | ||
|
|
||
| async def _merge_bot_messages_for_group( | ||
| self, group_id: str, messages: List[Any] | ||
| ) -> List[Any]: | ||
| """将数据库中的 Bot 回复按时间线合并进用户消息,供 few-shot 对话对提取。 | ||
|
|
||
| 实时学习链路已在调用前合并过 bot 消息(消息中已含 sender_id=='bot'), | ||
| 此时直接返回避免重复查询;批量/审批链路传入的原始消息只有用户发言, | ||
| 必须合并 bot 回复才能提取到 用户→bot 对话对(否则学习结果恒为空, | ||
| 即 issue #257 日志中的“未获得有效结果”)。 | ||
|
|
||
| 为避免跨窗口误配对:只取与这批用户消息时间窗口(末条向后一个宽容窗口) | ||
| 重叠的 Bot 回复;为避免“先 limit 后过滤”将有效回复滤空:先多取候选再在 | ||
| 内存过滤并截断到与用户消息相当的数量。 | ||
| """ | ||
| if not messages or not self.db_manager: | ||
| return messages | ||
| if any(self._msg_sender_id(m) == "bot" for m in messages): | ||
| return messages | ||
| try: | ||
| from sqlalchemy import select | ||
|
|
||
| from ...models.orm.message import BotMessage | ||
| from ..learning.sample_filter import should_ignore_learning_sample | ||
|
|
||
| user_ts = [ | ||
| self._msg_timestamp(m) | ||
| for m in messages | ||
| if self._msg_content(m) | ||
| ] | ||
| user_ts = [t for t in user_ts if t > 0] | ||
| if not user_ts: | ||
| return messages | ||
|
|
||
| forward_window_seconds = 3600.0 | ||
| t_lo = min(user_ts) | ||
| t_hi = max(user_ts) + forward_window_seconds | ||
| # 先多取候选,再过滤/截断,避免被 ignore 样本占满 limit 导致有效回复被漏掉 | ||
| fetch_limit = max(len(messages) * 3, 60) | ||
|
|
||
| async with self.db_manager.get_session() as session: | ||
| stmt = ( | ||
| select(BotMessage) | ||
| .where( | ||
| BotMessage.group_id == group_id, | ||
| BotMessage.timestamp >= t_lo, | ||
| BotMessage.timestamp <= t_hi, | ||
| ) | ||
| .order_by(BotMessage.timestamp.asc()) | ||
| .limit(fetch_limit) | ||
| ) | ||
| result = await session.execute(stmt) | ||
| bot_msgs: List[Dict[str, Any]] = [] | ||
| for row in result.scalars().all(): | ||
| if should_ignore_learning_sample( | ||
| row.message, sender_id="bot", is_bot=True | ||
| ): | ||
| continue | ||
| bot_msgs.append( | ||
| { | ||
| "sender_id": "bot", | ||
| "message": row.message, | ||
| "timestamp": float(row.timestamp), | ||
| } | ||
| ) | ||
| if len(bot_msgs) >= len(messages): | ||
| break | ||
| if not bot_msgs: | ||
| return messages | ||
| merged = list(messages) + bot_msgs | ||
| merged.sort(key=self._msg_timestamp) | ||
| return merged | ||
|
Comment on lines
+145
to
+214
Contributor
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. issue (bug_risk): The merge queries the latest BotMessage rows for the entire group and appends every retained reply to the supplied messages without restricting replies to the messages' timestamp range or pairing them with the corresponding user messages. A batch containing historical or sparse user messages therefore creates false user→bot pairs from unrelated conversations, and can also pair the final user message with a later bot reply. Triggers: When batch learning processes messages from a time window that does not exactly match the group's latest bot replies. Suggested fix: Restrict BotMessage rows to the relevant message time window and associate each reply with its preceding user message (or reuse the existing timeline-merging logic) before constructing pairs. |
||
| except Exception as exc: | ||
| logger.debug(f"合并 Bot 回复失败,使用原始消息继续: {exc}") | ||
| return messages | ||
|
|
||
| @classmethod | ||
| def get_instance(cls, config: PluginConfig = None, db_manager: DatabaseManager = None, context=None, llm_adapter=None) -> 'ExpressionPatternLearner': | ||
| """获取单例实例,支持延迟初始化""" | ||
|
|
@@ -184,6 +287,12 @@ async def trigger_learning_for_group( | |
| return False | ||
|
|
||
| try: | ||
| # 先合并数据库中已存储的 Bot 回复,才能提取到 用户→bot 对话对(修复批量链路恒为空) | ||
| recent_messages = await self._merge_bot_messages_for_group( | ||
| group_id, recent_messages | ||
| ) | ||
| if len(recent_messages) < 3: | ||
| return False | ||
| persona_id = normalize_persona_scope(persona_id) | ||
| user_scope = self._normalize_user_scope(user_id) | ||
| learning_scope = self._expression_scope_key(group_id, persona_id, user_scope) | ||
|
|
||
Uh oh!
There was an error while loading. Please reload this page.