Implement follow-up detection in dialog context to prevent context pollution in multi-topic conversations. Introduce is_likely_follow_up function to determine when to inject session history based on user input. Adjust _load_session_text2sql_context to conditionally include context based on follow-up status, enhancing SQL generation accuracy. Update relevant API endpoints to pass user text for improved context handling.
This commit is contained in:
Binary file not shown.
@@ -6,6 +6,7 @@ from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
import re
|
||||
from typing import Any, Dict, List, Tuple
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
@@ -14,6 +15,28 @@ _MAX_ASSISTANT_SQL_CHARS = 1400
|
||||
_MAX_ASSISTANT_TEXT_CHARS = 900
|
||||
_MAX_BLOCK_CHARS = 7500
|
||||
|
||||
# 续问/指代常见触发词:用于判定是否需要注入会话上文,避免多话题串扰。
|
||||
# 规则应偏保守:宁可把新话题当作“无上文”也不要带入无关上下文污染 SQL。
|
||||
_FOLLOW_UP_PATTERN = re.compile(
|
||||
r"(再|继续|同样|沿用|还是|照旧|刚才|上面|上一条|上一个|上次|上述|这个SQL|这条SQL|按上面|按刚才|在此基础上|同一口径|同口径|按之前|改成|改为|加上|加一下|筛一下|过滤一下|补充|补一下|追加|优化一下|调整一下|换成|换为)",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
|
||||
|
||||
def is_likely_follow_up(user_text: str) -> bool:
|
||||
"""
|
||||
仅基于用户本轮文本做“是否续问”的轻量判断。
|
||||
True -> 允许注入少量上文(最近 1~2 轮),用于指代消解/沿用口径
|
||||
False -> 新话题,禁用上文,避免上下文污染
|
||||
"""
|
||||
t = (user_text or "").strip()
|
||||
if not t:
|
||||
return False
|
||||
# 很短的“再来一个/同上/继续”等通常是续问
|
||||
if len(t) <= 10 and _FOLLOW_UP_PATTERN.search(t):
|
||||
return True
|
||||
return bool(_FOLLOW_UP_PATTERN.search(t))
|
||||
|
||||
|
||||
def _truncate(s: str, max_len: int) -> str:
|
||||
s = (s or "").strip()
|
||||
|
||||
Reference in New Issue
Block a user