Enhance environment configuration and error handling in API server and backend. Implement dynamic loading of .env files for various execution contexts, improve schema file path resolution, and refine vector search error handling in the orchestrator. Update ChromaDB integration to support memory mode and add persistence options for few-shot learning. Include additional logging for better traceability.

This commit is contained in:
lasean.zhou
2026-04-14 18:21:50 +08:00
parent ca8bc5e7de
commit 4ea3056e95
8 changed files with 161 additions and 41 deletions
+9 -7
View File
@@ -20,7 +20,7 @@ class SchemaIndexer:
功能:
1. 为表结构构建向量索引
2. 基于问题的表检索
3. 持久化存储和加载
3. Chroma 使用内存模式(EphemeralClient),进程退出后不保留;``persist_dir`` 仅作配置/统计引用
"""
def __init__(
@@ -34,17 +34,15 @@ class SchemaIndexer:
Args:
embedder: Embedding模型实例
persist_dir: 向量数据库持久化目录
persist_dir: 历史配置中的向量库路径(仅展示与统计,不落盘)
collection_name: 集合名称
"""
self.embedder = embedder
self.persist_dir = Path(persist_dir)
self.persist_dir.mkdir(parents=True, exist_ok=True)
self.collection_name = collection_name
# 初始化ChromaDB客户端
self.client = chromadb.PersistentClient(
path=str(self.persist_dir),
# 内存 Chroma:不落盘,每次进程需重新 build_index
self.client = chromadb.EphemeralClient(
settings=ChromaSettings(anonymized_telemetry=False),
)
@@ -54,7 +52,10 @@ class SchemaIndexer:
metadata={"hnsw:space": "cosine"}, # 使用余弦相似度
)
logger.info(f"[OK] 初始化SchemaIndexer: collection={self.collection_name}")
logger.info(
f"[OK] 初始化SchemaIndexer(Chroma内存): collection={self.collection_name}, "
f"配置路径引用={self.persist_dir}"
)
def build_index(
self,
@@ -249,4 +250,5 @@ class SchemaIndexer:
"total_columns": total_columns,
"avg_columns": total_columns / count if count > 0 else 0,
"persist_dir": str(self.persist_dir),
"chroma_mode": "memory",
}