Commit 6cb0ffa7 authored by Data Governance Dev's avatar Data Governance Dev

feat(backend): /api/dict/{tables,fields,all-table-names} 分页端点 + cache 复用

DataDictCache 之前只在 connect/test 写入;这次新增 3 个 GET 端点让前端按需切片:
- GET /api/dict/tables?db_type=&host=&port=&user=&database=&page=&page_size=&q=
  → 分页返回 table_summary,每项附加 field_count(前端不再算)
- GET /api/dict/fields?...&tables=t1,t2&page=&page_size=&q=
  → 分页返回 data_dictionary(按 tables 参数过滤),每项附加 table_comment
- GET /api/dict/all-table-names?...&q=
  → 仅返名字列表,供前端「全选 / 反选」按钮使用(不分页)

实现要点:
- 全部走 DataDictCache.get_instance().get(cfg),命中即 list[a:b] 切片(微秒级)
- cache miss 时现场抽一次(缺 password 抛 warning 让前端提示重连)
- 多连接并存按 5 元组 (db_type, host, port, database, user) 区分 key
- 排序:tables 按 table_name 字典序;fields 按 (table_name, ordinal_position)
  与旧前端 filteredFields 排序保持一致

踩坑:
- port=3306 默认参数必须放最末位(Python SyntaxError)
- list_dict_fields 重复定义 host 别名导致 FastAPI 参数解析歧义 → 简化为单一 host 参数
- list_columns SQL 不带 table_comment,/api/dict/fields 响应里要手动从 table_summary 关联
parent 21170b4b
......@@ -411,6 +411,218 @@ async def connect_test(req: TestConnectionRequest):
)
# ── 数据字典分页查询 ─────────────────────────────────────
# 2026-08-13 改造:connect/test 已经把全量字典写进 DataDictCache;
# 浏览器不再一次性把全量拖回前端,而是按 page/page_size + 筛选 q 切片返回。
# 实现要点:
# - cache 命中 → 直接 list[a:b] 切片(微秒级)
# - cache miss → 现场抽一次(要求 cfg.password 有效;缺失时返回空 + warning)
# - 多连接并存时按 5 元组 (db_type, host, port, database, user) 区分 key
# (前端当前只维护一个连接,但接口设计上仍以多连接为目标)
# - 所有响应都是 JSON dict,不引入新的 Pydantic 模型(前端用 dict 消费)
def _build_cfg_for_dict_endpoint(db_type: str, host: str, port: int,
user: str, database: str) -> DBConfig:
"""从 query 参数构造 DBConfig;password 留空 —— 分页端点不需要它。
cache key 不含 password(见 data_dict_cache._cache_key);cache 命中时
全程不连库,所以 password 缺省对分页查询没影响。cache miss 时才会真正连库,
此时 password 为空 → 抛错被外层 catch,返回 _warning。
"""
return DBConfig(
db_type=db_type, host=host, port=port,
user=user, password="", database=database,
charset="utf8mb4", connect_timeout=10,
)
def _resolve_dict_data(cfg: DBConfig) -> tuple[dict | None, str | None]:
"""从缓存取数据字典;miss 时现场抽一次。
Returns: (dict_data, warning)
- dict_data = None 且 warning 非空 → cache miss 且无法补抽(password 缺)
- dict_data = 有效 dict 且 warning = None → 正常
"""
cache = DataDictCache.get_instance()
cached = cache.get(cfg)
if cached is not None:
return cached, None
# cache miss —— password 缺,连库必失败
try:
from ..core.data_dict import extract_data_dictionary
dict_data = extract_data_dictionary(cfg)
cache.put(cfg, dict_data)
return dict_data, None
except Exception as e:
logger.warning(f"分页端点 cache miss 后现场抽字典失败: {e}")
return None, "cache miss 且无法重抽(请重新测试连接)"
def _filter_tables(tables: list[dict], q: str) -> list[dict]:
"""按 q 子串过滤 table_summary(表名 + 表注释,大小写不敏感)。"""
if not q:
return tables
ql = q.lower()
out = []
for t in tables:
if (ql in (t.get("table_name") or "").lower()
or ql in (t.get("table_comment") or "").lower()):
out.append(t)
return out
def _filter_fields(fields: list[dict], selected_tables: set[str], q: str) -> list[dict]:
"""按 selected_tables + q 子串过滤 data_dictionary。"""
out = fields
if selected_tables:
out = [f for f in out if f.get("table_name") in selected_tables]
if q:
ql = q.lower()
out = [f for f in out if
(ql in (f.get("column_name") or "").lower()
or ql in (f.get("column_comment") or "").lower())]
return out
@router.get("/dict/tables")
async def list_dict_tables(
db_type: str,
host: str,
port: int,
user: str,
database: str,
page: int = Query(1, ge=1),
page_size: int = Query(50, ge=1, le=500),
q: str = Query("", description="按表名 / 表注释子串筛选,大小写不敏感"),
):
"""分页返回缓存里的表级数据;每项附加 field_count(前端字段数列用)。"""
cfg = _build_cfg_for_dict_endpoint(db_type, host, port, user, database)
dict_data, warning = _resolve_dict_data(cfg)
if dict_data is None:
return {
"tables": [], "page": page, "page_size": page_size,
"total": 0, "has_more": False,
"_warning": warning,
}
tables_all = dict_data.get("table_summary", []) or []
# 计算每张表的字段数(前端浏览器左侧列表要显示「字段数」列)
# dict_data.data_dictionary 在缓存里已 dedupe,单次 O(N) 遍历即可
# (1971 张表 + 36212 字段 → 几 ms),不重复算
fields_all = dict_data.get("data_dictionary", []) or []
field_count: dict[str, int] = {}
for r in fields_all:
tn = r.get("table_name")
if tn:
field_count[tn] = field_count.get(tn, 0) + 1
tables_all = _filter_tables(tables_all, q)
# 按 table_name 排序 —— 首次进入浏览器看到的顺序稳定,便于跨页阅读
tables_all.sort(key=lambda t: (t.get("table_name") or "").lower())
total = len(tables_all)
start = (page - 1) * page_size
end = start + page_size
page_items = []
for t in tables_all[start:end]:
# 不修改原 dict(共享引用,污染缓存) —— 拷贝一份附加字段
item = dict(t)
item["field_count"] = field_count.get(item.get("table_name") or "", 0)
page_items.append(item)
return {
"tables": page_items,
"page": page,
"page_size": page_size,
"total": total,
"has_more": end < total,
}
@router.get("/dict/fields")
async def list_dict_fields(
db_type: str,
host: str,
user: str,
database: str,
port: int = 3306,
tables: str = Query("", description="逗号分隔的表名;空 = 不返回任何字段"),
page: int = Query(1, ge=1),
page_size: int = Query(50, ge=1, le=500),
q: str = Query("", description="按字段名 / 字段注释子串筛选"),
):
"""分页返回缓存里的字段数据,按 tables 过滤(仅返已选表的字段)。"""
cfg = _build_cfg_for_dict_endpoint(db_type, host, port, user, database)
dict_data, warning = _resolve_dict_data(cfg)
if dict_data is None:
return {
"fields": [], "page": page, "page_size": page_size,
"total": 0, "has_more": False,
"_warning": warning,
}
fields_all = dict_data.get("data_dictionary", []) or []
selected = {t.strip() for t in tables.split(",") if t.strip()}
if not selected:
# 没选表 = 当前 UI 文案「请在左侧勾选要分析的表」对应;返回空(不分页)
return {
"fields": [], "page": page, "page_size": page_size,
"total": 0, "has_more": False,
}
fields_all = _filter_fields(fields_all, selected, q)
# 按 (table_name, ordinal_position) 排序 —— 与旧前端 filteredFields 排序保持一致
fields_all.sort(key=lambda f: (
(f.get("table_name") or "").lower(),
f.get("ordinal_position") if isinstance(f.get("ordinal_position"), int) else 0,
))
total = len(fields_all)
start = (page - 1) * page_size
end = start + page_size
# 附加 table_comment(前端 addCustomRule 的字段要拿到;list_columns SQL 不带这列)
tables_all = dict_data.get("table_summary", []) or []
table_comment_map: dict[str, str] = {}
for t in tables_all:
tn = t.get("table_name")
if tn:
table_comment_map[tn] = t.get("table_comment") or ""
page_items = []
for f in fields_all[start:end]:
item = dict(f)
tn = item.get("table_name") or ""
if tn and "table_comment" not in item:
item["table_comment"] = table_comment_map.get(tn, "")
page_items.append(item)
return {
"fields": page_items,
"page": page,
"page_size": page_size,
"total": total,
"has_more": end < total,
}
@router.get("/dict/all-table-names")
async def list_all_table_names(
db_type: str,
host: str,
user: str,
database: str,
port: int = 3306,
q: str = Query("", description="按表名 / 表注释子串筛选"),
):
"""返回所有匹配 q 的表名(仅名字列表)—— 供前端「全选 / 反选」按钮使用。
全量遍历缓存里的表是几百微秒级(1971 张 in-memory),没有分页必要;
即便切到 10 万张表也只用一次 list comprehension。
"""
cfg = _build_cfg_for_dict_endpoint(db_type, host, port, user, database)
dict_data, warning = _resolve_dict_data(cfg)
if dict_data is None:
return {"names": [], "total": 0, "_warning": warning}
tables = dict_data.get("table_summary", []) or []
tables = _filter_tables(tables, q)
tables.sort(key=lambda t: (t.get("table_name") or "").lower())
return {
"names": [t.get("table_name") for t in tables],
"total": len(tables),
}
# ── 创建治理任务 ──
@router.post("/jobs", response_model=JobCreatedResponse)
async def create_job(req: ConnectRequest):
......
Markdown is supported
0%
or
You are about to add 0 people to the discussion. Proceed with caution.
Finish editing this message first!
Please register or to comment