Commit 8ac75cef authored by Data Governance Dev's avatar Data Governance Dev

feat(llm): 结构化 API 错误信息提取,429/401/403/500 等错误中文提示,前端实时日志友好展示

parent 9de8abeb
This diff is collapsed.
......@@ -61,8 +61,8 @@ def run_step2(dict_data: dict, llm: LLMClient | None = None,
# 3. 高频字段
if log:
log("INFO", "[3/3] 计算高频字段(出现 ≥10 张表)...", step="2")
redundancy = _find_redundancy(by_table)
log("INFO", "[3/3] 计算高频字段(出现 ≥10 张表,LLM 分类)...", step="2")
redundancy = _find_redundancy(by_table, llm, log)
if log:
log("INFO", f" · 高频字段 {len(redundancy)} 个", step="2")
......@@ -195,19 +195,97 @@ def _find_replacement(table: str, all_tables) -> str | None:
return None
def _find_redundancy(by_table: dict) -> list[dict]:
def _find_redundancy(by_table: dict, llm, log: Callable | None) -> list[dict]:
"""高频字段分析:Counter 预筛 → LLM 分类(必跑)。
Counter 找出出现 ≥10 张表的字段(纯规则、毫秒级)。
然后对每个候选调 LLM 分类:
- common_base 通用基础字段,可保留
- common_business 业务上合理共享,可保留
- suspicious 命名相同但含义可能不一致,需核对
- true_redundancy 真冗余,建议合并 / 抽字典
LLM 必跑:若调用失败(网络/解析)会让该字段标记 llm_failed,不影响其他字段;
若整批不可用由上层(orchestrator)视为关键失败。
"""
# 1. Counter 预筛
counter: Counter = Counter()
for cols in by_table.values():
type_by_field: dict[str, set[str]] = {}
tables_by_field: dict[str, set[str]] = {}
for tname, cols in by_table.items():
for c in cols:
counter[c["column_name"]] += 1
logger.debug(f"字段频次统计: 共 {len(counter)} 个不同字段名")
return [
fname = c["column_name"]
counter[fname] += 1
type_by_field.setdefault(fname, set()).add(c.get("data_type", "") or "")
tables_by_field.setdefault(fname, set()).add(tname)
candidates = [(f, n) for f, n in counter.most_common()
if n >= 10 and f not in COMMON_FIELDS]
logger.debug(f"字段频次统计: 共 {len(counter)} 个不同字段名, "
f"候选高频字段 {len(candidates)} 个")
if not candidates:
return []
# 2. 构造 LLM 输入
llm_inputs = [
{
"field": f,
"table_count": n,
"risk": "高频出现,需评估是否为业务必要字段" if n >= 30 else "中频出现",
"suggestion": "建议评审是否需要统一到公共字典表" if n >= 30 else "建议评审",
"sample_tables": sorted(tables_by_field[f])[:5],
"sample_types": sorted(type_by_field[f])[:3],
}
for f, n in counter.most_common()
if n >= 10 and f not in COMMON_FIELDS
for f, n in candidates
]
BATCH = 15
annotated: list[dict | None] = []
total_batches = (len(llm_inputs) + BATCH - 1) // BATCH
for batch_idx in range(0, len(llm_inputs), BATCH):
batch = llm_inputs[batch_idx:batch_idx + BATCH]
idx = batch_idx // BATCH + 1
if log:
log("INFO",
f" · LLM 分类批次 [{idx}/{total_batches}] "
f"({len(batch)} 个字段)",
step="2")
try:
results = llm.classify_redundant_fields_batch(batch)
except Exception as e:
if log:
log("ERROR",
f" · LLM 分类批次 [{idx}/{total_batches}] 整体失败: {e}(任务将终止)",
step="2")
logger.exception("Step 2 LLM 分类批次失败")
raise # required step → 抛给上层
annotated.extend(results)
# 3. 汇总:每个候选都给出最终记录(LLM 成功的带 classification / reasoning / recommendation;
# LLM 失败的标 llm_failed,仍保留 field + table_count 便于定位)
out: list[dict] = []
for cand, ann in zip(candidates, annotated):
f, n = cand
if ann:
out.append({
"field": f,
"table_count": n,
"classification": ann["classification"],
"reasoning": ann["reasoning"],
"recommendation": ann["recommendation"],
"source": "llm",
})
else:
out.append({
"field": f,
"table_count": n,
"classification": "unknown",
"reasoning": "LLM 解析失败",
"recommendation": "需人工核对",
"source": "llm_failed",
})
if log:
cls_count = {}
for r in out:
cls_count[r["classification"]] = cls_count.get(r["classification"], 0) + 1
summary = ", ".join(f"{k}={v}" for k, v in cls_count.items())
if log:
log("INFO", f" · LLM 分类结果: {summary}", step="2")
return out
\ No newline at end of file
"""Step 5: 缺失注释字段检查 + LLM 推测
"""Step 5: 缺失注释字段检查 + LLM 推测(必跑 LLM)
优先用 LLM 推测无注释字段的语义(替换原硬编码 COMMENT_HINTS);
LLM 不可用时降级为本地规则。
字段注释的语义判断本质上需要 LLM:
- 拼音首字母缩写(如 xzqhbm)、业务缩写、英文组合 → 必须由 LLM 翻译
- 硬编码字典兜底已删除(缺 LLM = 任务失败,无需 fallback)
产出:
- summary: 总数 + 推测命中率
- summary: 总数 + LLM 命中率
- by_table: 每表缺失注释数量
- predicted_comments: 推测出的注释(含置信度)
- unpredictable_sample: 未推测到的样本
- predicted_comments: LLM 推测出的注释(含置信度、来源 = "llm")
- unpredicted_sample: LLM 单条失败的样本(标 llm_failed)
"""
from __future__ import annotations
......@@ -21,31 +22,6 @@ from ..llm import LLMClient
logger = logging.getLogger(__name__)
# 兜底规则(LLM 不可用时使用)
FALLBACK_HINTS = {
"id": "主键ID",
"create_time": "创建时间",
"update_time": "更新时间",
"create_by": "创建人",
"update_by": "更新人",
"remark": "备注",
"del_flag": "删除标记",
"tenant_id": "租户ID",
"dept_id": "部门ID",
"project_id": "项目ID",
"project_name": "项目名称",
"project_code": "项目编号",
"site_id": "工地ID",
"site_name": "工地名称",
"status": "状态",
"sort_order": "排序",
"start_time": "开始时间",
"end_time": "结束时间",
"type": "类型",
"code": "编码",
}
def run_step5(dict_data: dict, llm: LLMClient | None = None,
log: Callable | None = None) -> dict:
columns = dict_data.get("data_dictionary", [])
......@@ -57,19 +33,16 @@ def run_step5(dict_data: dict, llm: LLMClient | None = None,
if log:
log("INFO", f"待检查字段总数: {len(columns)}", step="5")
# 1. 收集所有无注释字段,按 (table_name, table_comment) 分组
missing = []
by_table: dict[str, int] = defaultdict(int)
predicted = []
unpredictable = []
grouped: dict[tuple[str, str], list[dict]] = defaultdict(list)
# 先用规则快速批匹配
for r in columns:
comment = (r.get("column_comment") or "").strip()
if comment:
continue
by_table[r["table_name"]] += 1
fallback = FALLBACK_HINTS.get(r["column_name"])
entry = {
"table_name": r["table_name"],
"table_comment": r.get("table_comment", ""),
......@@ -81,73 +54,77 @@ def run_step5(dict_data: dict, llm: LLMClient | None = None,
"confidence": "low",
"reason": "",
}
if fallback:
entry["predicted"] = fallback
entry["confidence"] = "high"
entry["reason"] = "字段名匹配内置规则"
predicted.append(entry)
else:
unpredictable.append(entry)
missing.append(entry)
grouped[(r["table_name"], r.get("table_comment", ""))].append(entry)
if log:
log("INFO",
f" · 缺失注释字段: {len(missing)} 个 (覆盖 {len(by_table)} 张表)",
step="5")
log("INFO",
f" · 内置规则命中: {len(predicted)}, 需 LLM 推测: {len(unpredictable)}",
step="5")
# 用 LLM 处理 unpredictable(如果可用)
if llm and llm.available and unpredictable:
# 2. 按表分批调用 LLM(必跑;单批失败会让任务终止)
if not missing:
if log:
log("INFO", " · 无缺失注释字段,跳过 LLM", step="5")
return _empty_result(by_table)
BATCH = 15
predicted = []
unpredicted = []
llm_called = 0
for (tname, tcomment), fields in grouped.items():
# 把同表的字段切片成 BATCH 大小
chunks = [fields[i:i + BATCH] for i in range(0, len(fields), BATCH)]
for chunk_idx, chunk in enumerate(chunks, 1):
llm_called += 1
if log:
log("INFO", f"[LLM] 调用 LLM 推测 {len(unpredictable)} 个无规则命中的字段注释", step="5")
llm_predicted = []
still_unknown = []
for idx, entry in enumerate(unpredictable, 1):
log("INFO",
f" · LLM 推测 [{llm_called}] {tname} "
f"({chunk_idx}/{len(chunks)} 批, {len(chunk)} 个字段)",
step="5")
try:
r = llm.predict_field_comment(
table_name=entry["table_name"],
table_comment=entry["table_comment"],
column_name=entry["column_name"],
data_type=entry["data_type"],
results = llm.predict_field_comments_batch(
table_name=tname,
table_comment=tcomment,
fields=[{"column_name": e["column_name"],
"data_type": e["data_type"]} for e in chunk],
)
except Exception as e:
# 单批失败 → 让任务终止(required step)
if log:
log("ERROR",
f" · LLM 推测失败({tname} 第 {chunk_idx} 批): {e}(任务将终止)",
step="5")
logger.exception("Step 5 LLM 推测批次失败")
raise
# 把 LLM 结果写回 entry
for entry, r in zip(chunk, results):
if r and r.get("comment"):
entry["predicted"] = r["comment"]
entry["confidence"] = r.get("confidence", "low")
entry["reason"] = "LLM 推测"
llm_predicted.append(entry)
# 从 predicted 列表的视角也算推测成功
entry["source"] = "llm"
predicted.append(entry)
if log and idx % 5 == 0:
log("DEBUG",
f" · LLM 推测进度 {idx}/{len(unpredictable)} "
f"(已成功 {len(llm_predicted)})",
step="5")
else:
still_unknown.append(entry)
except Exception as e:
if log:
log("WARN",
f"LLM 推测失败 ({entry['table_name']}.{entry['column_name']}): {e}",
step="5")
still_unknown.append(entry)
if log:
log("INFO", f"[LLM] 推测成功 {len(llm_predicted)} / {len(unpredictable)}", step="5")
unpredictable = still_unknown
entry["reason"] = "LLM 解析失败"
entry["source"] = "llm_failed"
unpredicted.append(entry)
# 按表聚合
# 3. 按表聚合
by_table_list = sorted(
[{"table_name": t, "missing_count": c} for t, c in by_table.items()],
key=lambda x: x["missing_count"], reverse=True
)[:20]
if log:
log("INFO",
f" · LLM 命中 {len(predicted)}/{len(missing)}, "
f"LLM 解析失败 {len(unpredicted)}",
step="5")
log("INFO",
f"缺失注释 {len(missing)} 个字段,覆盖 {len(by_table)} 张表;"
f"推测成功 {len(predicted)}(规则 {sum(1 for p in predicted if p.get('reason') == '字段名匹配内置规则')}, "
f"LLM {sum(1 for p in predicted if p.get('reason') == 'LLM 推测')})",
f"LLM 推测成功 {len(predicted)}",
step="5")
return {
......@@ -155,11 +132,27 @@ def run_step5(dict_data: dict, llm: LLMClient | None = None,
"total_missing_comments": len(missing),
"tables_affected": len(by_table),
"predicted_count": len(predicted),
"predicted_by_rules": sum(1 for p in predicted if p.get("reason") == "字段名匹配内置规则"),
"predicted_by_llm": sum(1 for p in predicted if p.get("reason") == "LLM 推测"),
"unpredicted_count": len(unpredictable),
"predicted_by_llm": len(predicted),
"unpredicted_count": len(unpredicted),
"llm_calls": llm_called,
},
"by_table": by_table_list,
"predicted_comments": predicted[:50],
"unpredictable_sample": unpredictable[:50],
"unpredictable_sample": unpredicted[:50],
}
def _empty_result(by_table: dict) -> dict:
return {
"summary": {
"total_missing_comments": 0,
"tables_affected": len(by_table),
"predicted_count": 0,
"predicted_by_llm": 0,
"unpredicted_count": 0,
"llm_calls": 0,
},
"by_table": [],
"predicted_comments": [],
"unpredictable_sample": [],
}
\ No newline at end of file
Markdown is supported
0%
or
You are about to add 0 people to the discussion. Proceed with caution.
Finish editing this message first!
Please register or to comment