Enhance impact analysis and configuration for vector persistence and API integration. Update main.spec to clarify data handling for ChromaDB, ensuring vector directories are excluded from packaging. Modify Text2SQLOrchestrator to unify question normalization for consistent SQL generation across languages. Introduce SchemaIndexer improvements for persistent vector storage and optimize embedding retrieval processes. Update documentation and comments for clarity on configuration changes and behavior adjustments.

This commit is contained in:
陈辅元
2026-04-15 11:25:19 +08:00
parent 2d0b1ab36f
commit a55fd18915
15 changed files with 318 additions and 50 deletions
+10 -15
View File
@@ -66,7 +66,9 @@ class Text2SQLOrchestrator:
vector_db_path: 向量数据库路径
max_retry: 最大重试次数(包含首次生成)
use_vector_search: 是否使用向量检索粗筛
translate_english_to_zh: 无中日韩字符的英文问句是否先译为中文再走检索与生成
translate_english_to_zh: 为 True 时(默认)对**所有**非空问句做一次 LLM 归一(temperature=0),
输出一句标准中文供检索与生成;使同一语义的中英文表述对齐,从而 SQL 一致。为 False 时
不做归一(原样英文/中文)。环境变量 ``TRANSLATE_EN_TO_ZH=false`` 可关闭。
"""
self.schema_manager = schema_manager
self.max_retry = max_retry
@@ -120,7 +122,7 @@ class Text2SQLOrchestrator:
f"[OK] Text2SQLOrchestrator初始化完成: "
f"max_retry={max_retry}, use_vector_search={use_vector_search}"
+ (f", fewshot=on" if self.fewshot_enabled else "")
+ (", en→zh=on" if self.translate_english_to_zh else ", en→zh=off")
+ (", nl→zh_norm=on" if self.translate_english_to_zh else ", nl→zh_norm=off")
)
@staticmethod
@@ -174,12 +176,7 @@ class Text2SQLOrchestrator:
try:
indexer = self._get_vector_index()
# 确保索引已构建
if indexer.count() == 0:
logger.info("向量索引为空,正在构建...")
indexer.build_index(self.schema_manager, force_rebuild=True)
logger.info(f"[OK] 索引构建完成,共 {indexer.count()} 张表")
indexer.ensure_index_for_schema(self.schema_manager)
# 检索
logger.info(f"[Orchestrator] 开始向量检索: query='{question[:50]}...'")
@@ -585,26 +582,24 @@ class Text2SQLOrchestrator:
Returns:
GenerationResult对象
"""
from utils.question_locale import looks_like_english_only
original_question = (question or "").strip()
translation_meta: Dict = {}
work_question = original_question
if self.translate_english_to_zh and looks_like_english_only(original_question):
if self.translate_english_to_zh and original_question:
try:
zh = self.deepseek.translate_nl_question_to_zh(original_question).strip()
zh = self.deepseek.normalize_nl_question_for_text2sql(original_question).strip()
if zh and len(zh) >= 2:
work_question = zh
translation_meta["question_original"] = original_question
translation_meta["question_zh_normalized"] = zh
logger.info(
"[GEN] 英文已译为中文:%s",
"[GEN] 问句已归一中文:%s",
zh[:120] + ("…" if len(zh) > 120 else ""),
)
else:
logger.warning("[GEN] 英译中结果为空或过短,使用原文")
logger.warning("[GEN] 归一结果为空或过短,使用原文")
except Exception as e:
logger.warning("[GEN] 英译中失败,使用原文: %s", e)
logger.warning("[GEN] 问句归一失败,使用原文: %s", e)
question = work_question
dc_raw = (dialog_context or "").strip()