Commit 8ac75cef authored by Data Governance Dev's avatar Data Governance Dev

feat(llm): 结构化 API 错误信息提取,429/401/403/500 等错误中文提示,前端实时日志友好展示

parent 9de8abeb
...@@ -219,12 +219,12 @@ class LLMClient: ...@@ -219,12 +219,12 @@ class LLMClient:
# ── 核心调用 ── # ── 核心调用 ──
def complete(self, prompt: str, system: str = "", json_mode: bool = False) -> str: def complete(self, prompt: str, system: str = "", json_mode: bool = False) -> str:
"""调用 LLM,返回纯文本。失败抛 LLMUnavailable。""" """调用 LLM,返回纯文本。失败抛 LLMUnavailable(携带可读的错误信息)。"""
if not self.available: if not self.available:
raise LLMUnavailable("LLM 客户端未配置 API Key") raise LLMUnavailable("LLM 客户端未配置 API Key")
try: try:
if self.cfg.provider == "anthropic": if self.cfg.provider in ("anthropic", "minimax"):
kwargs: dict[str, Any] = { kwargs: dict[str, Any] = {
"model": self.cfg.model, "model": self.cfg.model,
"max_tokens": self.cfg.max_tokens, "max_tokens": self.cfg.max_tokens,
...@@ -248,9 +248,20 @@ class LLMClient: ...@@ -248,9 +248,20 @@ class LLMClient:
kwargs["response_format"] = {"type": "json_object"} kwargs["response_format"] = {"type": "json_object"}
resp = self._provider.chat.completions.create(**kwargs) resp = self._provider.chat.completions.create(**kwargs)
return resp.choices[0].message.content return resp.choices[0].message.content
# 未支持的 provider——按理 _init_provider 已拦过,这里再防一次
raise LLMUnavailable(
f"complete() 未实现 provider={self.cfg.provider!r} 的调用分支"
)
except LLMUnavailable:
raise
except Exception as e: except Exception as e:
logger.warning(f"LLM 调用失败: {e}") detail = _format_api_error(e, self.cfg.provider)
raise LLMUnavailable(str(e)) logger.warning(
f"LLM 调用失败 (provider={self.cfg.provider}, "
f"model={self.cfg.effective_model}): {detail}"
)
raise LLMUnavailable(detail)
# ── JSON 模式(带解析兜底) ── # ── JSON 模式(带解析兜底) ──
def complete_json(self, prompt: str, system: str = "") -> dict | list | None: def complete_json(self, prompt: str, system: str = "") -> dict | list | None:
...@@ -261,37 +272,141 @@ class LLMClient: ...@@ -261,37 +272,141 @@ class LLMClient:
text = self.complete(prompt, system=json_system, json_mode=True) text = self.complete(prompt, system=json_system, json_mode=True)
return _safe_parse_json(text) return _safe_parse_json(text)
# ── 业务方法:字段注释推测 ── # ── 业务方法:字段注释推测(批处理,required by Step 5) ──
def predict_field_comment( def predict_field_comments_batch(
self, self,
table_name: str, table_name: str,
table_comment: str, table_comment: str,
column_name: str, fields: list[dict],
data_type: str, ) -> list[dict | None]:
) -> dict | None: """批量推测字段语义,返回与输入等长的 list,None 表示该项失败。
"""推测字段语义,返回 {"comment": str, "confidence": "high|medium|low"} 或 None"""
prompt = f"""你是数据库治理专家。请根据下面的信息,推测一个字段的中文注释。
表名:{table_name} Input fields: [{"column_name": str, "data_type": str}, ...]
表注释:{table_comment or '(无)'} Output: [{"column_name": str, "comment": str,
字段名:{column_name} "confidence": "high|medium|low"}, ...] 与输入同序
数据类型:{data_type} """
if not fields:
return []
if not self.available:
raise LLMUnavailable("LLM 客户端未配置 API Key")
fields_json = "\n".join(
f'{i+1}. {f["column_name"]} ({f.get("data_type", "")})'
for i, f in enumerate(fields)
)
prompt = f"""你是数据库治理专家。表名 `{table_name}`({table_comment or '无注释'})下有以下 {len(fields)} 个无注释字段,请逐个推测中文注释。
字段清单:
{fields_json}
要求: 要求:
1. 给出 5-20 字的中文注释 1. 每条给出 5-20 字的中文注释
2. 如字段名是拼音首字母缩写(如 xzqhbm),翻译为中文 2. 拼音首字母缩写(如 xzqhbm)翻译为中文
3. 如字段名是英文组合(如 create_time),翻译为中文 3. 英文组合(如 create_time)翻译为中文
4. 给出置信度(high=非常确定;medium=较确定;low=猜测) 4. 给出置信度(high=非常确定;medium=较确定;low=猜测)
5. 严格保持输入顺序,返回数组长度 == {len(fields)}
严格返回 JSON 格式:{{"comment": "...", "confidence": "high|medium|low"}} 严格返回 JSON 数组(不要用对象包裹):
[
{{"column_name": "<原字段名>", "comment": "<中文注释>", "confidence": "high|medium|low"}},
...
]
""" """
result = self.complete_json(prompt) result = self.complete_json(prompt)
if isinstance(result, dict) and "comment" in result: if not isinstance(result, list):
return { logger.warning(f"批量注释推测返回非数组: {type(result).__name__}")
"comment": str(result["comment"]).strip(), return [None] * len(fields)
"confidence": result.get("confidence", "low"),
} # 按 column_name 对齐回输入顺序(容错:LLM 可能打乱顺序)
return None by_name = {r.get("column_name"): r for r in result if isinstance(r, dict)}
out: list[dict | None] = []
for f in fields:
r = by_name.get(f["column_name"])
if r and r.get("comment"):
out.append({
"column_name": f["column_name"],
"comment": str(r["comment"]).strip(),
"confidence": r.get("confidence", "low"),
})
else:
out.append(None)
return out
# ── 业务方法:冗余字段分类(批处理,required by Step 2) ──
def classify_redundant_fields_batch(
self,
fields: list[dict],
) -> list[dict | None]:
"""批量分类高频字段是否为真冗余,返回与输入等长的 list。
Input fields: [{
"field": str, # 字段名
"table_count": int, # 出现表数
"sample_tables": [str], # 抽样表名(最多 5 张)
"sample_types": [str], # 抽样的数据类型列表
}, ...]
Output: [{
"field": str,
"classification": "common_base|common_business|suspicious|true_redundancy",
"reasoning": str,
"recommendation": str,
}, ...] 与输入同序
"""
if not fields:
return []
if not self.available:
raise LLMUnavailable("LLM 客户端未配置 API Key")
fields_json = "\n".join(
f'{i+1}. {f["field"]}(出现在 {f.get("table_count", "?")} 张表,'
f'类型 {",".join(f.get("sample_types", [])[:3]) or "?"},'
f'抽样表 {", ".join(f.get("sample_tables", [])[:5])})'
for i, f in enumerate(fields)
)
prompt = f"""数据库治理专家:以下 {len(fields)} 个字段在数据库中出现频次较高,请判断它们的"高频出现"是否合理。
字段清单:
{fields_json}
分类维度(互斥):
- "common_base" :通用基础字段,所有表都该有(如 id / create_time / update_time / create_by)
- "common_business" :业务上合理共享(如 status / sort_order / type / code),可以保留
- "suspicious" :命名相同但含义可能不一致,需要核对每个表的具体定义
- "true_redundancy" :明确冗余——同名同义却被多表各自维护,应该抽公共字典或合并
要求:
1. 严格保持输入顺序,返回数组长度 == {len(fields)}
2. reasoning 1-2 句说明判断依据
3. recommendation 给出整改建议(保留/合并/抽字典/核对)
严格返回 JSON 数组:
[
{{"field": "<原字段名>", "classification": "common_base|common_business|suspicious|true_redundancy",
"reasoning": "...", "recommendation": "..."}},
...
]
"""
result = self.complete_json(prompt)
if not isinstance(result, list):
logger.warning(f"批量冗余分类返回非数组: {type(result).__name__}")
return [None] * len(fields)
by_name = {r.get("field"): r for r in result if isinstance(r, dict)}
out: list[dict | None] = []
for f in fields:
r = by_name.get(f["field"])
if r and r.get("classification") in (
"common_base", "common_business", "suspicious", "true_redundancy"
):
out.append({
"field": f["field"],
"classification": r["classification"],
"reasoning": str(r.get("reasoning", "")).strip(),
"recommendation": str(r.get("recommendation", "")).strip(),
})
else:
out.append(None)
return out
# ── 业务方法:表合并建议 ── # ── 业务方法:表合并建议 ──
def suggest_table_merge( def suggest_table_merge(
...@@ -388,7 +503,114 @@ class LLMClient: ...@@ -388,7 +503,114 @@ class LLMClient:
return None return None
# ── JSON 解析兜底 ──────────────────────────────────────── # ── API 错误信息提取 ──────────────────────────────────────
def _format_api_error(exc: Exception, provider: str) -> str:
"""从 SDK 异常中提取可读的错误信息,供前端实时日志展示。
优先处理 Anthropic/OpenAI SDK 的结构化异常,
兜底处理网络/超时等通用异常。
"""
exc_type = type(exc).__name__
exc_msg = str(exc)
# ── Anthropic SDK 异常(MiniMax 兼容接口也走这里) ──
try:
from anthropic import (
APIStatusError,
RateLimitError,
AuthenticationError,
PermissionDeniedError,
NotFoundError,
APIConnectionError,
APITimeoutError,
)
if isinstance(exc, RateLimitError):
return (
f"API Error: 请求被限流 (429) · {_extract_body_message(exc) or exc_msg}"
)
if isinstance(exc, AuthenticationError):
return (
f"API Error: 认证失败 (401) · 请检查 API Key 是否正确或已过期"
)
if isinstance(exc, PermissionDeniedError):
return (
f"API Error: 权限不足 (403) · {_extract_body_message(exc) or exc_msg}"
)
if isinstance(exc, NotFoundError):
return (
f"API Error: 资源不存在 (404) · {_extract_body_message(exc) or exc_msg}"
)
if isinstance(exc, APIStatusError):
status = getattr(exc, "status_code", "?")
body_msg = _extract_body_message(exc)
detail = body_msg or exc_msg
# 对常见状态码给出中文提示
hint = {
429: "请升级 Token Plan 套餐或稍后重试",
500: "服务端内部错误,请稍后重试",
502: "网关错误,服务可能暂时不可用",
503: "服务暂不可用,请稍后重试",
}.get(status, "")
if hint:
return f"API Error: 请求被拒绝 ({status}) · {detail} ({hint})"
return f"API Error: 请求失败 ({status}) · {detail}"
if isinstance(exc, APIConnectionError):
return f"API Error: 网络连接失败 · {exc_msg}"
if isinstance(exc, APITimeoutError):
return f"API Error: 请求超时 · {exc_msg}"
except ImportError:
pass
# ── OpenAI SDK 异常 ──
try:
from openai import (
APIStatusError as OAIStatusError,
RateLimitError as OAIRateLimitError,
AuthenticationError as OAIAuthError,
APIConnectionError as OAIConnectionError,
APITimeoutError as OAITimeoutError,
)
if isinstance(exc, OAIRateLimitError):
return (
f"API Error: 请求被限流 (429) · {exc_msg}"
)
if isinstance(exc, OAIAuthError):
return (
f"API Error: 认证失败 (401) · 请检查 API Key 是否正确或已过期"
)
if isinstance(exc, OAIStatusError):
status = getattr(exc, "status_code", "?")
return f"API Error: 请求失败 ({status}) · {exc_msg}"
if isinstance(exc, OAIConnectionError):
return f"API Error: 网络连接失败 · {exc_msg}"
if isinstance(exc, OAITimeoutError):
return f"API Error: 请求超时 · {exc_msg}"
except ImportError:
pass
# ── 通用网络/超时异常 ──
import builtins
if isinstance(exc, builtins.ConnectionError):
return f"API Error: 网络连接失败 · {exc_msg}"
if isinstance(exc, builtins.TimeoutError):
return f"API Error: 请求超时 · {exc_msg}"
# ── 兜底 ──
return f"API Error: {exc_type} · {exc_msg}"
def _extract_body_message(exc: Exception) -> str | None:
"""尝试从 Anthropic APIStatusError 的 body 中提取错误详情。"""
body = getattr(exc, "body", None)
if not body or not isinstance(body, dict):
return None
# body 结构: {"error": {"message": "..."}}
error = body.get("error")
if isinstance(error, dict):
msg = error.get("message")
if msg:
return str(msg)
return None
def _safe_parse_json(text: str) -> Any: def _safe_parse_json(text: str) -> Any:
"""从 LLM 返回中提取 JSON(容忍代码块包裹、尾部多余文字)""" """从 LLM 返回中提取 JSON(容忍代码块包裹、尾部多余文字)"""
if not text: if not text:
......
...@@ -61,8 +61,8 @@ def run_step2(dict_data: dict, llm: LLMClient | None = None, ...@@ -61,8 +61,8 @@ def run_step2(dict_data: dict, llm: LLMClient | None = None,
# 3. 高频字段 # 3. 高频字段
if log: if log:
log("INFO", "[3/3] 计算高频字段(出现 ≥10 张表)...", step="2") log("INFO", "[3/3] 计算高频字段(出现 ≥10 张表,LLM 分类)...", step="2")
redundancy = _find_redundancy(by_table) redundancy = _find_redundancy(by_table, llm, log)
if log: if log:
log("INFO", f" · 高频字段 {len(redundancy)} 个", step="2") log("INFO", f" · 高频字段 {len(redundancy)} 个", step="2")
...@@ -195,19 +195,97 @@ def _find_replacement(table: str, all_tables) -> str | None: ...@@ -195,19 +195,97 @@ def _find_replacement(table: str, all_tables) -> str | None:
return None return None
def _find_redundancy(by_table: dict) -> list[dict]: def _find_redundancy(by_table: dict, llm, log: Callable | None) -> list[dict]:
"""高频字段分析:Counter 预筛 → LLM 分类(必跑)。
Counter 找出出现 ≥10 张表的字段(纯规则、毫秒级)。
然后对每个候选调 LLM 分类:
- common_base 通用基础字段,可保留
- common_business 业务上合理共享,可保留
- suspicious 命名相同但含义可能不一致,需核对
- true_redundancy 真冗余,建议合并 / 抽字典
LLM 必跑:若调用失败(网络/解析)会让该字段标记 llm_failed,不影响其他字段;
若整批不可用由上层(orchestrator)视为关键失败。
"""
# 1. Counter 预筛
counter: Counter = Counter() counter: Counter = Counter()
for cols in by_table.values(): type_by_field: dict[str, set[str]] = {}
tables_by_field: dict[str, set[str]] = {}
for tname, cols in by_table.items():
for c in cols: for c in cols:
counter[c["column_name"]] += 1 fname = c["column_name"]
logger.debug(f"字段频次统计: 共 {len(counter)} 个不同字段名") counter[fname] += 1
return [ type_by_field.setdefault(fname, set()).add(c.get("data_type", "") or "")
tables_by_field.setdefault(fname, set()).add(tname)
candidates = [(f, n) for f, n in counter.most_common()
if n >= 10 and f not in COMMON_FIELDS]
logger.debug(f"字段频次统计: 共 {len(counter)} 个不同字段名, "
f"候选高频字段 {len(candidates)} 个")
if not candidates:
return []
# 2. 构造 LLM 输入
llm_inputs = [
{ {
"field": f, "field": f,
"table_count": n, "table_count": n,
"risk": "高频出现,需评估是否为业务必要字段" if n >= 30 else "中频出现", "sample_tables": sorted(tables_by_field[f])[:5],
"suggestion": "建议评审是否需要统一到公共字典表" if n >= 30 else "建议评审", "sample_types": sorted(type_by_field[f])[:3],
} }
for f, n in counter.most_common() for f, n in candidates
if n >= 10 and f not in COMMON_FIELDS
] ]
BATCH = 15
annotated: list[dict | None] = []
total_batches = (len(llm_inputs) + BATCH - 1) // BATCH
for batch_idx in range(0, len(llm_inputs), BATCH):
batch = llm_inputs[batch_idx:batch_idx + BATCH]
idx = batch_idx // BATCH + 1
if log:
log("INFO",
f" · LLM 分类批次 [{idx}/{total_batches}] "
f"({len(batch)} 个字段)",
step="2")
try:
results = llm.classify_redundant_fields_batch(batch)
except Exception as e:
if log:
log("ERROR",
f" · LLM 分类批次 [{idx}/{total_batches}] 整体失败: {e}(任务将终止)",
step="2")
logger.exception("Step 2 LLM 分类批次失败")
raise # required step → 抛给上层
annotated.extend(results)
# 3. 汇总:每个候选都给出最终记录(LLM 成功的带 classification / reasoning / recommendation;
# LLM 失败的标 llm_failed,仍保留 field + table_count 便于定位)
out: list[dict] = []
for cand, ann in zip(candidates, annotated):
f, n = cand
if ann:
out.append({
"field": f,
"table_count": n,
"classification": ann["classification"],
"reasoning": ann["reasoning"],
"recommendation": ann["recommendation"],
"source": "llm",
})
else:
out.append({
"field": f,
"table_count": n,
"classification": "unknown",
"reasoning": "LLM 解析失败",
"recommendation": "需人工核对",
"source": "llm_failed",
})
if log:
cls_count = {}
for r in out:
cls_count[r["classification"]] = cls_count.get(r["classification"], 0) + 1
summary = ", ".join(f"{k}={v}" for k, v in cls_count.items())
if log:
log("INFO", f" · LLM 分类结果: {summary}", step="2")
return out
\ No newline at end of file
"""Step 5: 缺失注释字段检查 + LLM 推测 """Step 5: 缺失注释字段检查 + LLM 推测(必跑 LLM)
优先用 LLM 推测无注释字段的语义(替换原硬编码 COMMENT_HINTS); 字段注释的语义判断本质上需要 LLM:
LLM 不可用时降级为本地规则。 - 拼音首字母缩写(如 xzqhbm)、业务缩写、英文组合 → 必须由 LLM 翻译
- 硬编码字典兜底已删除(缺 LLM = 任务失败,无需 fallback)
产出: 产出:
- summary: 总数 + 推测命中率 - summary: 总数 + LLM 命中率
- by_table: 每表缺失注释数量 - by_table: 每表缺失注释数量
- predicted_comments: 推测出的注释(含置信度) - predicted_comments: LLM 推测出的注释(含置信度、来源 = "llm")
- unpredictable_sample: 未推测到的样本 - unpredicted_sample: LLM 单条失败的样本(标 llm_failed)
""" """
from __future__ import annotations from __future__ import annotations
...@@ -21,31 +22,6 @@ from ..llm import LLMClient ...@@ -21,31 +22,6 @@ from ..llm import LLMClient
logger = logging.getLogger(__name__) logger = logging.getLogger(__name__)
# 兜底规则(LLM 不可用时使用)
FALLBACK_HINTS = {
"id": "主键ID",
"create_time": "创建时间",
"update_time": "更新时间",
"create_by": "创建人",
"update_by": "更新人",
"remark": "备注",
"del_flag": "删除标记",
"tenant_id": "租户ID",
"dept_id": "部门ID",
"project_id": "项目ID",
"project_name": "项目名称",
"project_code": "项目编号",
"site_id": "工地ID",
"site_name": "工地名称",
"status": "状态",
"sort_order": "排序",
"start_time": "开始时间",
"end_time": "结束时间",
"type": "类型",
"code": "编码",
}
def run_step5(dict_data: dict, llm: LLMClient | None = None, def run_step5(dict_data: dict, llm: LLMClient | None = None,
log: Callable | None = None) -> dict: log: Callable | None = None) -> dict:
columns = dict_data.get("data_dictionary", []) columns = dict_data.get("data_dictionary", [])
...@@ -57,19 +33,16 @@ def run_step5(dict_data: dict, llm: LLMClient | None = None, ...@@ -57,19 +33,16 @@ def run_step5(dict_data: dict, llm: LLMClient | None = None,
if log: if log:
log("INFO", f"待检查字段总数: {len(columns)}", step="5") log("INFO", f"待检查字段总数: {len(columns)}", step="5")
# 1. 收集所有无注释字段,按 (table_name, table_comment) 分组
missing = [] missing = []
by_table: dict[str, int] = defaultdict(int) by_table: dict[str, int] = defaultdict(int)
predicted = [] grouped: dict[tuple[str, str], list[dict]] = defaultdict(list)
unpredictable = []
# 先用规则快速批匹配
for r in columns: for r in columns:
comment = (r.get("column_comment") or "").strip() comment = (r.get("column_comment") or "").strip()
if comment: if comment:
continue continue
by_table[r["table_name"]] += 1 by_table[r["table_name"]] += 1
fallback = FALLBACK_HINTS.get(r["column_name"])
entry = { entry = {
"table_name": r["table_name"], "table_name": r["table_name"],
"table_comment": r.get("table_comment", ""), "table_comment": r.get("table_comment", ""),
...@@ -81,73 +54,77 @@ def run_step5(dict_data: dict, llm: LLMClient | None = None, ...@@ -81,73 +54,77 @@ def run_step5(dict_data: dict, llm: LLMClient | None = None,
"confidence": "low", "confidence": "low",
"reason": "", "reason": "",
} }
if fallback:
entry["predicted"] = fallback
entry["confidence"] = "high"
entry["reason"] = "字段名匹配内置规则"
predicted.append(entry)
else:
unpredictable.append(entry)
missing.append(entry) missing.append(entry)
grouped[(r["table_name"], r.get("table_comment", ""))].append(entry)
if log: if log:
log("INFO", log("INFO",
f" · 缺失注释字段: {len(missing)} 个 (覆盖 {len(by_table)} 张表)", f" · 缺失注释字段: {len(missing)} 个 (覆盖 {len(by_table)} 张表)",
step="5") step="5")
log("INFO",
f" · 内置规则命中: {len(predicted)}, 需 LLM 推测: {len(unpredictable)}",
step="5")
# 用 LLM 处理 unpredictable(如果可用) # 2. 按表分批调用 LLM(必跑;单批失败会让任务终止)
if llm and llm.available and unpredictable: if not missing:
if log:
log("INFO", " · 无缺失注释字段,跳过 LLM", step="5")
return _empty_result(by_table)
BATCH = 15
predicted = []
unpredicted = []
llm_called = 0
for (tname, tcomment), fields in grouped.items():
# 把同表的字段切片成 BATCH 大小
chunks = [fields[i:i + BATCH] for i in range(0, len(fields), BATCH)]
for chunk_idx, chunk in enumerate(chunks, 1):
llm_called += 1
if log: if log:
log("INFO", f"[LLM] 调用 LLM 推测 {len(unpredictable)} 个无规则命中的字段注释", step="5") log("INFO",
llm_predicted = [] f" · LLM 推测 [{llm_called}] {tname} "
still_unknown = [] f"({chunk_idx}/{len(chunks)} 批, {len(chunk)} 个字段)",
for idx, entry in enumerate(unpredictable, 1): step="5")
try: try:
r = llm.predict_field_comment( results = llm.predict_field_comments_batch(
table_name=entry["table_name"], table_name=tname,
table_comment=entry["table_comment"], table_comment=tcomment,
column_name=entry["column_name"], fields=[{"column_name": e["column_name"],
data_type=entry["data_type"], "data_type": e["data_type"]} for e in chunk],
) )
except Exception as e:
# 单批失败 → 让任务终止(required step)
if log:
log("ERROR",
f" · LLM 推测失败({tname} 第 {chunk_idx} 批): {e}(任务将终止)",
step="5")
logger.exception("Step 5 LLM 推测批次失败")
raise
# 把 LLM 结果写回 entry
for entry, r in zip(chunk, results):
if r and r.get("comment"): if r and r.get("comment"):
entry["predicted"] = r["comment"] entry["predicted"] = r["comment"]
entry["confidence"] = r.get("confidence", "low") entry["confidence"] = r.get("confidence", "low")
entry["reason"] = "LLM 推测" entry["reason"] = "LLM 推测"
llm_predicted.append(entry) entry["source"] = "llm"
# 从 predicted 列表的视角也算推测成功
predicted.append(entry) predicted.append(entry)
if log and idx % 5 == 0:
log("DEBUG",
f" · LLM 推测进度 {idx}/{len(unpredictable)} "
f"(已成功 {len(llm_predicted)})",
step="5")
else: else:
still_unknown.append(entry) entry["reason"] = "LLM 解析失败"
except Exception as e: entry["source"] = "llm_failed"
if log: unpredicted.append(entry)
log("WARN",
f"LLM 推测失败 ({entry['table_name']}.{entry['column_name']}): {e}",
step="5")
still_unknown.append(entry)
if log:
log("INFO", f"[LLM] 推测成功 {len(llm_predicted)} / {len(unpredictable)}", step="5")
unpredictable = still_unknown
# 按表聚合 # 3. 按表聚合
by_table_list = sorted( by_table_list = sorted(
[{"table_name": t, "missing_count": c} for t, c in by_table.items()], [{"table_name": t, "missing_count": c} for t, c in by_table.items()],
key=lambda x: x["missing_count"], reverse=True key=lambda x: x["missing_count"], reverse=True
)[:20] )[:20]
if log: if log:
log("INFO",
f" · LLM 命中 {len(predicted)}/{len(missing)}, "
f"LLM 解析失败 {len(unpredicted)}",
step="5")
log("INFO", log("INFO",
f"缺失注释 {len(missing)} 个字段,覆盖 {len(by_table)} 张表;" f"缺失注释 {len(missing)} 个字段,覆盖 {len(by_table)} 张表;"
f"推测成功 {len(predicted)}(规则 {sum(1 for p in predicted if p.get('reason') == '字段名匹配内置规则')}, " f"LLM 推测成功 {len(predicted)}",
f"LLM {sum(1 for p in predicted if p.get('reason') == 'LLM 推测')})",
step="5") step="5")
return { return {
...@@ -155,11 +132,27 @@ def run_step5(dict_data: dict, llm: LLMClient | None = None, ...@@ -155,11 +132,27 @@ def run_step5(dict_data: dict, llm: LLMClient | None = None,
"total_missing_comments": len(missing), "total_missing_comments": len(missing),
"tables_affected": len(by_table), "tables_affected": len(by_table),
"predicted_count": len(predicted), "predicted_count": len(predicted),
"predicted_by_rules": sum(1 for p in predicted if p.get("reason") == "字段名匹配内置规则"), "predicted_by_llm": len(predicted),
"predicted_by_llm": sum(1 for p in predicted if p.get("reason") == "LLM 推测"), "unpredicted_count": len(unpredicted),
"unpredicted_count": len(unpredictable), "llm_calls": llm_called,
}, },
"by_table": by_table_list, "by_table": by_table_list,
"predicted_comments": predicted[:50], "predicted_comments": predicted[:50],
"unpredictable_sample": unpredictable[:50], "unpredictable_sample": unpredicted[:50],
}
def _empty_result(by_table: dict) -> dict:
return {
"summary": {
"total_missing_comments": 0,
"tables_affected": len(by_table),
"predicted_count": 0,
"predicted_by_llm": 0,
"unpredicted_count": 0,
"llm_calls": 0,
},
"by_table": [],
"predicted_comments": [],
"unpredictable_sample": [],
} }
\ No newline at end of file
Markdown is supported
0%
or
You are about to add 0 people to the discussion. Proceed with caution.
Finish editing this message first!
Please register or to comment