Enhance environment configuration and error handling in API server and backend. Implement dynamic loading of .env files for various execution contexts, improve schema file path resolution, and refine vector search error handling in the orchestrator. Update ChromaDB integration to support memory mode and add persistence options for few-shot learning. Include additional logging for better traceability.
This commit is contained in:
@@ -20,7 +20,7 @@ class SchemaIndexer:
|
||||
功能:
|
||||
1. 为表结构构建向量索引
|
||||
2. 基于问题的表检索
|
||||
3. 持久化存储和加载
|
||||
3. Chroma 使用内存模式(EphemeralClient),进程退出后不保留;``persist_dir`` 仅作配置/统计引用
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
@@ -34,17 +34,15 @@ class SchemaIndexer:
|
||||
|
||||
Args:
|
||||
embedder: Embedding模型实例
|
||||
persist_dir: 向量数据库持久化目录
|
||||
persist_dir: 历史配置中的向量库路径(仅展示与统计,不落盘)
|
||||
collection_name: 集合名称
|
||||
"""
|
||||
self.embedder = embedder
|
||||
self.persist_dir = Path(persist_dir)
|
||||
self.persist_dir.mkdir(parents=True, exist_ok=True)
|
||||
self.collection_name = collection_name
|
||||
|
||||
# 初始化ChromaDB客户端
|
||||
self.client = chromadb.PersistentClient(
|
||||
path=str(self.persist_dir),
|
||||
# 内存 Chroma:不落盘,每次进程需重新 build_index
|
||||
self.client = chromadb.EphemeralClient(
|
||||
settings=ChromaSettings(anonymized_telemetry=False),
|
||||
)
|
||||
|
||||
@@ -54,7 +52,10 @@ class SchemaIndexer:
|
||||
metadata={"hnsw:space": "cosine"}, # 使用余弦相似度
|
||||
)
|
||||
|
||||
logger.info(f"[OK] 初始化SchemaIndexer: collection={self.collection_name}")
|
||||
logger.info(
|
||||
f"[OK] 初始化SchemaIndexer(Chroma内存): collection={self.collection_name}, "
|
||||
f"配置路径引用={self.persist_dir}"
|
||||
)
|
||||
|
||||
def build_index(
|
||||
self,
|
||||
@@ -249,4 +250,5 @@ class SchemaIndexer:
|
||||
"total_columns": total_columns,
|
||||
"avg_columns": total_columns / count if count > 0 else 0,
|
||||
"persist_dir": str(self.persist_dir),
|
||||
"chroma_mode": "memory",
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user