Commit 8ac75cef authored by Data Governance Dev's avatar Data Governance Dev

feat(llm): 结构化 API 错误信息提取,429/401/403/500 等错误中文提示,前端实时日志友好展示

parent 9de8abeb
This diff is collapsed.
...@@ -61,8 +61,8 @@ def run_step2(dict_data: dict, llm: LLMClient | None = None, ...@@ -61,8 +61,8 @@ def run_step2(dict_data: dict, llm: LLMClient | None = None,
# 3. 高频字段 # 3. 高频字段
if log: if log:
log("INFO", "[3/3] 计算高频字段(出现 ≥10 张表)...", step="2") log("INFO", "[3/3] 计算高频字段(出现 ≥10 张表,LLM 分类)...", step="2")
redundancy = _find_redundancy(by_table) redundancy = _find_redundancy(by_table, llm, log)
if log: if log:
log("INFO", f" · 高频字段 {len(redundancy)} 个", step="2") log("INFO", f" · 高频字段 {len(redundancy)} 个", step="2")
...@@ -195,19 +195,97 @@ def _find_replacement(table: str, all_tables) -> str | None: ...@@ -195,19 +195,97 @@ def _find_replacement(table: str, all_tables) -> str | None:
return None return None
def _find_redundancy(by_table: dict) -> list[dict]: def _find_redundancy(by_table: dict, llm, log: Callable | None) -> list[dict]:
"""高频字段分析:Counter 预筛 → LLM 分类(必跑)。
Counter 找出出现 ≥10 张表的字段(纯规则、毫秒级)。
然后对每个候选调 LLM 分类:
- common_base 通用基础字段,可保留
- common_business 业务上合理共享,可保留
- suspicious 命名相同但含义可能不一致,需核对
- true_redundancy 真冗余,建议合并 / 抽字典
LLM 必跑:若调用失败(网络/解析)会让该字段标记 llm_failed,不影响其他字段;
若整批不可用由上层(orchestrator)视为关键失败。
"""
# 1. Counter 预筛
counter: Counter = Counter() counter: Counter = Counter()
for cols in by_table.values(): type_by_field: dict[str, set[str]] = {}
tables_by_field: dict[str, set[str]] = {}
for tname, cols in by_table.items():
for c in cols: for c in cols:
counter[c["column_name"]] += 1 fname = c["column_name"]
logger.debug(f"字段频次统计: 共 {len(counter)} 个不同字段名") counter[fname] += 1
return [ type_by_field.setdefault(fname, set()).add(c.get("data_type", "") or "")
tables_by_field.setdefault(fname, set()).add(tname)
candidates = [(f, n) for f, n in counter.most_common()
if n >= 10 and f not in COMMON_FIELDS]
logger.debug(f"字段频次统计: 共 {len(counter)} 个不同字段名, "
f"候选高频字段 {len(candidates)} 个")
if not candidates:
return []
# 2. 构造 LLM 输入
llm_inputs = [
{ {
"field": f, "field": f,
"table_count": n, "table_count": n,
"risk": "高频出现,需评估是否为业务必要字段" if n >= 30 else "中频出现", "sample_tables": sorted(tables_by_field[f])[:5],
"suggestion": "建议评审是否需要统一到公共字典表" if n >= 30 else "建议评审", "sample_types": sorted(type_by_field[f])[:3],
} }
for f, n in counter.most_common() for f, n in candidates
if n >= 10 and f not in COMMON_FIELDS ]
] BATCH = 15
\ No newline at end of file annotated: list[dict | None] = []
total_batches = (len(llm_inputs) + BATCH - 1) // BATCH
for batch_idx in range(0, len(llm_inputs), BATCH):
batch = llm_inputs[batch_idx:batch_idx + BATCH]
idx = batch_idx // BATCH + 1
if log:
log("INFO",
f" · LLM 分类批次 [{idx}/{total_batches}] "
f"({len(batch)} 个字段)",
step="2")
try:
results = llm.classify_redundant_fields_batch(batch)
except Exception as e:
if log:
log("ERROR",
f" · LLM 分类批次 [{idx}/{total_batches}] 整体失败: {e}(任务将终止)",
step="2")
logger.exception("Step 2 LLM 分类批次失败")
raise # required step → 抛给上层
annotated.extend(results)
# 3. 汇总:每个候选都给出最终记录(LLM 成功的带 classification / reasoning / recommendation;
# LLM 失败的标 llm_failed,仍保留 field + table_count 便于定位)
out: list[dict] = []
for cand, ann in zip(candidates, annotated):
f, n = cand
if ann:
out.append({
"field": f,
"table_count": n,
"classification": ann["classification"],
"reasoning": ann["reasoning"],
"recommendation": ann["recommendation"],
"source": "llm",
})
else:
out.append({
"field": f,
"table_count": n,
"classification": "unknown",
"reasoning": "LLM 解析失败",
"recommendation": "需人工核对",
"source": "llm_failed",
})
if log:
cls_count = {}
for r in out:
cls_count[r["classification"]] = cls_count.get(r["classification"], 0) + 1
summary = ", ".join(f"{k}={v}" for k, v in cls_count.items())
if log:
log("INFO", f" · LLM 分类结果: {summary}", step="2")
return out
\ No newline at end of file
"""Step 5: 缺失注释字段检查 + LLM 推测 """Step 5: 缺失注释字段检查 + LLM 推测(必跑 LLM)
优先用 LLM 推测无注释字段的语义(替换原硬编码 COMMENT_HINTS); 字段注释的语义判断本质上需要 LLM:
LLM 不可用时降级为本地规则。 - 拼音首字母缩写(如 xzqhbm)、业务缩写、英文组合 → 必须由 LLM 翻译
- 硬编码字典兜底已删除(缺 LLM = 任务失败,无需 fallback)
产出: 产出:
- summary: 总数 + 推测命中率 - summary: 总数 + LLM 命中率
- by_table: 每表缺失注释数量 - by_table: 每表缺失注释数量
- predicted_comments: 推测出的注释(含置信度) - predicted_comments: LLM 推测出的注释(含置信度、来源 = "llm")
- unpredictable_sample: 未推测到的样本 - unpredicted_sample: LLM 单条失败的样本(标 llm_failed)
""" """
from __future__ import annotations from __future__ import annotations
...@@ -21,31 +22,6 @@ from ..llm import LLMClient ...@@ -21,31 +22,6 @@ from ..llm import LLMClient
logger = logging.getLogger(__name__) logger = logging.getLogger(__name__)
# 兜底规则(LLM 不可用时使用)
FALLBACK_HINTS = {
"id": "主键ID",
"create_time": "创建时间",
"update_time": "更新时间",
"create_by": "创建人",
"update_by": "更新人",
"remark": "备注",
"del_flag": "删除标记",
"tenant_id": "租户ID",
"dept_id": "部门ID",
"project_id": "项目ID",
"project_name": "项目名称",
"project_code": "项目编号",
"site_id": "工地ID",
"site_name": "工地名称",
"status": "状态",
"sort_order": "排序",
"start_time": "开始时间",
"end_time": "结束时间",
"type": "类型",
"code": "编码",
}
def run_step5(dict_data: dict, llm: LLMClient | None = None, def run_step5(dict_data: dict, llm: LLMClient | None = None,
log: Callable | None = None) -> dict: log: Callable | None = None) -> dict:
columns = dict_data.get("data_dictionary", []) columns = dict_data.get("data_dictionary", [])
...@@ -57,19 +33,16 @@ def run_step5(dict_data: dict, llm: LLMClient | None = None, ...@@ -57,19 +33,16 @@ def run_step5(dict_data: dict, llm: LLMClient | None = None,
if log: if log:
log("INFO", f"待检查字段总数: {len(columns)}", step="5") log("INFO", f"待检查字段总数: {len(columns)}", step="5")
# 1. 收集所有无注释字段,按 (table_name, table_comment) 分组
missing = [] missing = []
by_table: dict[str, int] = defaultdict(int) by_table: dict[str, int] = defaultdict(int)
predicted = [] grouped: dict[tuple[str, str], list[dict]] = defaultdict(list)
unpredictable = []
# 先用规则快速批匹配
for r in columns: for r in columns:
comment = (r.get("column_comment") or "").strip() comment = (r.get("column_comment") or "").strip()
if comment: if comment:
continue continue
by_table[r["table_name"]] += 1 by_table[r["table_name"]] += 1
fallback = FALLBACK_HINTS.get(r["column_name"])
entry = { entry = {
"table_name": r["table_name"], "table_name": r["table_name"],
"table_comment": r.get("table_comment", ""), "table_comment": r.get("table_comment", ""),
...@@ -81,73 +54,77 @@ def run_step5(dict_data: dict, llm: LLMClient | None = None, ...@@ -81,73 +54,77 @@ def run_step5(dict_data: dict, llm: LLMClient | None = None,
"confidence": "low", "confidence": "low",
"reason": "", "reason": "",
} }
if fallback:
entry["predicted"] = fallback
entry["confidence"] = "high"
entry["reason"] = "字段名匹配内置规则"
predicted.append(entry)
else:
unpredictable.append(entry)
missing.append(entry) missing.append(entry)
grouped[(r["table_name"], r.get("table_comment", ""))].append(entry)
if log: if log:
log("INFO", log("INFO",
f" · 缺失注释字段: {len(missing)} 个 (覆盖 {len(by_table)} 张表)", f" · 缺失注释字段: {len(missing)} 个 (覆盖 {len(by_table)} 张表)",
step="5") step="5")
log("INFO",
f" · 内置规则命中: {len(predicted)}, 需 LLM 推测: {len(unpredictable)}",
step="5")
# 用 LLM 处理 unpredictable(如果可用) # 2. 按表分批调用 LLM(必跑;单批失败会让任务终止)
if llm and llm.available and unpredictable: if not missing:
if log: if log:
log("INFO", f"[LLM] 调用 LLM 推测 {len(unpredictable)} 个无规则命中的字段注释", step="5") log("INFO", " · 无缺失注释字段,跳过 LLM", step="5")
llm_predicted = [] return _empty_result(by_table)
still_unknown = []
for idx, entry in enumerate(unpredictable, 1): BATCH = 15
predicted = []
unpredicted = []
llm_called = 0
for (tname, tcomment), fields in grouped.items():
# 把同表的字段切片成 BATCH 大小
chunks = [fields[i:i + BATCH] for i in range(0, len(fields), BATCH)]
for chunk_idx, chunk in enumerate(chunks, 1):
llm_called += 1
if log:
log("INFO",
f" · LLM 推测 [{llm_called}] {tname} "
f"({chunk_idx}/{len(chunks)} 批, {len(chunk)} 个字段)",
step="5")
try: try:
r = llm.predict_field_comment( results = llm.predict_field_comments_batch(
table_name=entry["table_name"], table_name=tname,
table_comment=entry["table_comment"], table_comment=tcomment,
column_name=entry["column_name"], fields=[{"column_name": e["column_name"],
data_type=entry["data_type"], "data_type": e["data_type"]} for e in chunk],
) )
except Exception as e:
# 单批失败 → 让任务终止(required step)
if log:
log("ERROR",
f" · LLM 推测失败({tname} 第 {chunk_idx} 批): {e}(任务将终止)",
step="5")
logger.exception("Step 5 LLM 推测批次失败")
raise
# 把 LLM 结果写回 entry
for entry, r in zip(chunk, results):
if r and r.get("comment"): if r and r.get("comment"):
entry["predicted"] = r["comment"] entry["predicted"] = r["comment"]
entry["confidence"] = r.get("confidence", "low") entry["confidence"] = r.get("confidence", "low")
entry["reason"] = "LLM 推测" entry["reason"] = "LLM 推测"
llm_predicted.append(entry) entry["source"] = "llm"
# 从 predicted 列表的视角也算推测成功
predicted.append(entry) predicted.append(entry)
if log and idx % 5 == 0:
log("DEBUG",
f" · LLM 推测进度 {idx}/{len(unpredictable)} "
f"(已成功 {len(llm_predicted)})",
step="5")
else: else:
still_unknown.append(entry) entry["reason"] = "LLM 解析失败"
except Exception as e: entry["source"] = "llm_failed"
if log: unpredicted.append(entry)
log("WARN",
f"LLM 推测失败 ({entry['table_name']}.{entry['column_name']}): {e}",
step="5")
still_unknown.append(entry)
if log:
log("INFO", f"[LLM] 推测成功 {len(llm_predicted)} / {len(unpredictable)}", step="5")
unpredictable = still_unknown
# 按表聚合 # 3. 按表聚合
by_table_list = sorted( by_table_list = sorted(
[{"table_name": t, "missing_count": c} for t, c in by_table.items()], [{"table_name": t, "missing_count": c} for t, c in by_table.items()],
key=lambda x: x["missing_count"], reverse=True key=lambda x: x["missing_count"], reverse=True
)[:20] )[:20]
if log: if log:
log("INFO",
f" · LLM 命中 {len(predicted)}/{len(missing)}, "
f"LLM 解析失败 {len(unpredicted)}",
step="5")
log("INFO", log("INFO",
f"缺失注释 {len(missing)} 个字段,覆盖 {len(by_table)} 张表;" f"缺失注释 {len(missing)} 个字段,覆盖 {len(by_table)} 张表;"
f"推测成功 {len(predicted)}(规则 {sum(1 for p in predicted if p.get('reason') == '字段名匹配内置规则')}, " f"LLM 推测成功 {len(predicted)}",
f"LLM {sum(1 for p in predicted if p.get('reason') == 'LLM 推测')})",
step="5") step="5")
return { return {
...@@ -155,11 +132,27 @@ def run_step5(dict_data: dict, llm: LLMClient | None = None, ...@@ -155,11 +132,27 @@ def run_step5(dict_data: dict, llm: LLMClient | None = None,
"total_missing_comments": len(missing), "total_missing_comments": len(missing),
"tables_affected": len(by_table), "tables_affected": len(by_table),
"predicted_count": len(predicted), "predicted_count": len(predicted),
"predicted_by_rules": sum(1 for p in predicted if p.get("reason") == "字段名匹配内置规则"), "predicted_by_llm": len(predicted),
"predicted_by_llm": sum(1 for p in predicted if p.get("reason") == "LLM 推测"), "unpredicted_count": len(unpredicted),
"unpredicted_count": len(unpredictable), "llm_calls": llm_called,
}, },
"by_table": by_table_list, "by_table": by_table_list,
"predicted_comments": predicted[:50], "predicted_comments": predicted[:50],
"unpredictable_sample": unpredictable[:50], "unpredictable_sample": unpredicted[:50],
}
def _empty_result(by_table: dict) -> dict:
return {
"summary": {
"total_missing_comments": 0,
"tables_affected": len(by_table),
"predicted_count": 0,
"predicted_by_llm": 0,
"unpredicted_count": 0,
"llm_calls": 0,
},
"by_table": [],
"predicted_comments": [],
"unpredictable_sample": [],
} }
\ No newline at end of file
Markdown is supported
0%
or
You are about to add 0 people to the discussion. Proceed with caution.
Finish editing this message first!
Please register or to comment