320 lines
11 KiB
Python
320 lines
11 KiB
Python
"""
|
||
Schema加载器 - 解析JSON/DDL格式的数据库Schema
|
||
"""
|
||
|
||
import json
|
||
import re
|
||
from pathlib import Path
|
||
from typing import List, Dict, Optional, Tuple
|
||
import logging
|
||
|
||
from .models import Table, Column, ForeignKey, DatabaseSchema
|
||
|
||
logger = logging.getLogger(__name__)
|
||
|
||
|
||
class SchemaLoader:
|
||
"""Schema加载器 - 支持JSON和DDL格式"""
|
||
|
||
def __init__(self, schema_dir: str = None):
|
||
"""
|
||
初始化Schema加载器
|
||
|
||
Args:
|
||
schema_dir: Schema文件目录路径
|
||
"""
|
||
self.schema_dir = Path(schema_dir) if schema_dir else None
|
||
|
||
def load_from_json(
|
||
self, json_path: str, g3sb_meta_path: Optional[str] = None
|
||
) -> DatabaseSchema:
|
||
"""
|
||
从JSON文件加载Schema
|
||
|
||
JSON格式示例:
|
||
{
|
||
"database": "db_name",
|
||
"tables": [
|
||
{
|
||
"name": "table1",
|
||
"comment": "表描述",
|
||
"columns": [
|
||
{"name": "col1", "type": "INT", "comment": "...", "nullable": false}
|
||
],
|
||
"primary_keys": ["col1"],
|
||
"foreign_keys": [
|
||
{"columns": ["col2"], "ref_table": "table2", "ref_columns": ["col1"]}
|
||
]
|
||
}
|
||
]
|
||
}
|
||
|
||
亦支持 G3SB table_structure.json:顶层为 ``schemas`` 字典(与 ``tables`` 数组二选一
|
||
时优先使用非空的 ``schemas``)。可选 ``g3sb_meta_path`` 指向 table_meta.json
|
||
以合并表级注释。
|
||
|
||
Args:
|
||
json_path: JSON文件路径
|
||
g3sb_meta_path: 可选,G3SB table_meta.json(含 tables.{表名}.comment)
|
||
|
||
Returns:
|
||
DatabaseSchema对象
|
||
"""
|
||
path = Path(json_path)
|
||
if not path.exists():
|
||
raise FileNotFoundError(f"Schema文件不存在: {json_path}")
|
||
|
||
with open(path, 'r', encoding='utf-8') as f:
|
||
data = json.load(f)
|
||
|
||
# 提取数据库名
|
||
db_name = data.get("database", path.stem)
|
||
|
||
schemas_map = data.get("schemas")
|
||
tables_data = data.get("tables")
|
||
|
||
# 优先非空 schemas(G3SB 结构字典),避免误把其它 truthy 的 tables 键当好列表解析
|
||
if isinstance(schemas_map, dict) and schemas_map:
|
||
table_meta: Dict = {}
|
||
if g3sb_meta_path:
|
||
mp = Path(g3sb_meta_path)
|
||
if mp.is_file():
|
||
with open(mp, 'r', encoding='utf-8') as mf:
|
||
meta_data = json.load(mf)
|
||
table_meta = meta_data.get("tables", {}) or {}
|
||
else:
|
||
logger.warning(
|
||
"G3SB meta 文件不存在,将跳过表注释: %s", g3sb_meta_path
|
||
)
|
||
|
||
tables = []
|
||
for table_name, structure_str in schemas_map.items():
|
||
if not isinstance(structure_str, str):
|
||
continue
|
||
comment = None
|
||
if table_meta and isinstance(table_meta.get(table_name), dict):
|
||
comment = table_meta[table_name].get("comment")
|
||
columns = self._parse_g3sb_table_structure(structure_str)
|
||
tables.append(
|
||
Table(
|
||
name=table_name,
|
||
comment=comment,
|
||
columns=columns,
|
||
primary_keys=self._extract_primary_keys(columns),
|
||
foreign_keys=[],
|
||
)
|
||
)
|
||
logger.info(
|
||
"[OK] 加载Schema完成(G3SB schemas): %s, 共%d张表", db_name, len(tables)
|
||
)
|
||
return DatabaseSchema(name=db_name, tables=tables)
|
||
|
||
if isinstance(tables_data, list) and tables_data:
|
||
tables = [self._parse_table(table_data) for table_data in tables_data]
|
||
logger.info(f"[OK] 加载Schema完成: {db_name}, 共{len(tables)}张表")
|
||
return DatabaseSchema(name=db_name, tables=tables)
|
||
|
||
tables = []
|
||
logger.info(f"[OK] 加载Schema完成: {db_name}, 共{len(tables)}张表")
|
||
return DatabaseSchema(name=db_name, tables=tables)
|
||
|
||
def load_from_g3sb_format(self, meta_path: str, structure_path: str) -> DatabaseSchema:
|
||
"""
|
||
加载G3SB系统的数据字典格式(两个JSON文件)
|
||
|
||
Args:
|
||
meta_path: table_meta.json路径(表注释)
|
||
structure_path: table_structure.json路径(表结构)
|
||
|
||
Returns:
|
||
DatabaseSchema对象
|
||
"""
|
||
# 1. 加载表元数据(表注释)
|
||
with open(meta_path, 'r', encoding='utf-8') as f:
|
||
meta_data = json.load(f)
|
||
table_meta = meta_data.get("tables", {})
|
||
|
||
# 2. 加载表结构
|
||
with open(structure_path, 'r', encoding='utf-8') as f:
|
||
structure_data = json.load(f)
|
||
table_structures = structure_data.get("schemas", {})
|
||
|
||
# 3. 解析所有表
|
||
tables = []
|
||
for table_name, structure_str in table_structures.items():
|
||
comment = table_meta.get(table_name, {}).get("comment")
|
||
|
||
# 解析表结构字符串
|
||
# 格式: "TABLE TableName (col1:TYPE -- comment, col2:TYPE, ...)"
|
||
columns = self._parse_g3sb_table_structure(structure_str)
|
||
|
||
table = Table(
|
||
name=table_name,
|
||
comment=comment,
|
||
columns=columns,
|
||
primary_keys=self._extract_primary_keys(columns),
|
||
foreign_keys=[] # G3SB格式没有外键信息,后续可补充
|
||
)
|
||
tables.append(table)
|
||
|
||
logger.info(f"[OK] 加载G3SB Schema完成: 共{len(tables)}张表")
|
||
return DatabaseSchema(name="G3SB_DB", tables=tables)
|
||
|
||
def _parse_table(self, data: Dict) -> Table:
|
||
"""解析单表JSON数据"""
|
||
columns = []
|
||
for col_data in data.get("columns", []):
|
||
col = Column(
|
||
name=col_data["name"],
|
||
data_type=col_data["type"],
|
||
comment=col_data.get("comment"),
|
||
nullable=col_data.get("nullable", True),
|
||
is_primary_key=col_data.get("is_primary_key", False),
|
||
)
|
||
columns.append(col)
|
||
|
||
# 外键解析
|
||
foreign_keys = []
|
||
for fk_data in data.get("foreign_keys", []):
|
||
fk = ForeignKey(
|
||
columns=fk_data["columns"],
|
||
ref_table=fk_data["ref_table"],
|
||
ref_columns=fk_data["ref_columns"],
|
||
)
|
||
foreign_keys.append(fk)
|
||
|
||
return Table(
|
||
name=data["name"],
|
||
comment=data.get("comment"),
|
||
columns=columns,
|
||
primary_keys=data.get("primary_keys", []),
|
||
foreign_keys=foreign_keys,
|
||
)
|
||
|
||
def _parse_g3sb_table_structure(self, structure_str: str) -> List[Column]:
|
||
"""
|
||
解析G3SB格式的表结构字符串
|
||
|
||
示例:
|
||
"TABLE BCAccountCash (AccountID:NCHAR, RegionID:NCHAR, CurrencyID:NCHAR, Settled:DECIMAL -- Settled balance, ...)"
|
||
"""
|
||
# 提取括号内的内容
|
||
match = re.search(r'\((.*)\)', structure_str)
|
||
if not match:
|
||
logger.warning(f"无法解析表结构: {structure_str[:100]}")
|
||
return []
|
||
|
||
inner = match.group(1)
|
||
|
||
columns = []
|
||
# 按逗号分割字段(注意注释中可能包含逗号)
|
||
parts = self._split_columns(inner)
|
||
|
||
for part in parts:
|
||
part = part.strip()
|
||
if not part:
|
||
continue
|
||
|
||
# 解析字段定义:name:TYPE [-- comment]
|
||
# 支持格式: "FieldName:DATATYPE" 或 "FieldName:DATATYPE -- comment"
|
||
col_match = re.match(r'^(\w+)\s*:\s*([A-Za-z0-9()]+)', part)
|
||
if not col_match:
|
||
continue
|
||
|
||
col_name = col_match.group(1).strip()
|
||
col_type = col_match.group(2).strip()
|
||
|
||
# 提取注释(ASCII -- 或 G3SB 常用的 Unicode 长破折号 — U+2014)
|
||
comment = None
|
||
comment_match = re.search(r'(?:--|\u2014)\s*(.+)', part)
|
||
if comment_match:
|
||
comment = comment_match.group(1).strip()
|
||
|
||
# 判断是否可为空(通常有默认值或未标注NOT NULL即为NULL)
|
||
nullable = True # G3SB格式默认允许NULL
|
||
|
||
column = Column(
|
||
name=col_name,
|
||
data_type=col_type,
|
||
comment=comment,
|
||
nullable=nullable,
|
||
)
|
||
columns.append(column)
|
||
|
||
return columns
|
||
|
||
def _split_columns(self, inner: str) -> List[str]:
|
||
"""
|
||
按字段边界拆分(G3SB 注释里常有英文逗号,且用 — 而非 --)。
|
||
|
||
仅在「后面紧跟 标识符: 」的逗号处切分,这样注释内的逗号不会误拆列。
|
||
"""
|
||
if not inner or not inner.strip():
|
||
return []
|
||
# 下一列以 Name:TYPE 开头;避免在括号嵌套里误匹配可再收紧(当前 G3SB 类型无顶层逗号)
|
||
parts = re.split(r",\s*(?=\w+\s*:)", inner)
|
||
return [p.strip() for p in parts if p.strip()]
|
||
|
||
def _extract_primary_keys(self, columns: List[Column]) -> List[str]:
|
||
"""从字段列表中提取主键(简单启发式:字段名包含ID或明确标记)"""
|
||
pk_candidates = []
|
||
for col in columns:
|
||
# 简单规则:字段名以ID结尾,或名称包含key/id
|
||
if col.name.upper().endswith('ID') or 'KEY' in col.name.upper():
|
||
pk_candidates.append(col.name)
|
||
return pk_candidates[:1] # 暂时只取一个主键(简化)
|
||
|
||
def load_all_schemas(self) -> List[DatabaseSchema]:
|
||
"""
|
||
加载schema_dir下的所有Schema文件
|
||
|
||
Returns:
|
||
DatabaseSchema列表
|
||
"""
|
||
if not self.schema_dir:
|
||
raise ValueError("未指定schema_dir")
|
||
|
||
schemas = []
|
||
for json_file in self.schema_dir.glob("*.json"):
|
||
try:
|
||
schema = self.load_from_json(str(json_file))
|
||
schemas.append(schema)
|
||
except Exception as e:
|
||
logger.error(f"加载Schema失败 {json_file}: {e}")
|
||
|
||
logger.info(f"[OK] 共加载{len(schemas)}个Schema")
|
||
return schemas
|
||
|
||
|
||
# 便捷函数
|
||
def load_schema_from_g3sb(meta_path: str, structure_path: str) -> DatabaseSchema:
|
||
"""
|
||
从G3SB格式加载Schema的便捷函数
|
||
|
||
Args:
|
||
meta_path: table_meta.json路径
|
||
structure_path: table_structure.json路径
|
||
|
||
Returns:
|
||
DatabaseSchema对象
|
||
"""
|
||
loader = SchemaLoader()
|
||
return loader.load_from_g3sb_format(meta_path, structure_path)
|
||
|
||
|
||
def load_schema_from_json(
|
||
json_path: str, g3sb_meta_path: Optional[str] = None
|
||
) -> DatabaseSchema:
|
||
"""
|
||
从标准JSON加载Schema的便捷函数
|
||
|
||
Args:
|
||
json_path: JSON文件路径
|
||
g3sb_meta_path: 可选,G3SB table_meta.json
|
||
|
||
Returns:
|
||
DatabaseSchema对象
|
||
"""
|
||
loader = SchemaLoader()
|
||
return loader.load_from_json(json_path, g3sb_meta_path=g3sb_meta_path)
|