Skip to content
Projects
Groups
Snippets
Help
Loading...
Help
Support
Keyboard shortcuts
?
Submit feedback
Contribute to GitLab
Sign in
Toggle navigation
D
db-tools
Project overview
Project overview
Details
Activity
Releases
Repository
Repository
Files
Commits
Branches
Tags
Contributors
Graph
Compare
Issues
0
Issues
0
List
Boards
Labels
Milestones
Merge Requests
0
Merge Requests
0
CI / CD
CI / CD
Pipelines
Jobs
Schedules
Analytics
Analytics
CI / CD
Repository
Value Stream
Wiki
Wiki
Snippets
Snippets
Members
Members
Collapse sidebar
Close sidebar
Activity
Graph
Create a new issue
Jobs
Commits
Issue Boards
Open sidebar
yzy
db-tools
Commits
8ac75cef
Commit
8ac75cef
authored
Aug 05, 2026
by
Data Governance Dev
Browse files
Options
Browse Files
Download
Email Patches
Plain Diff
feat(llm): 结构化 API 错误信息提取,429/401/403/500 等错误中文提示,前端实时日志友好展示
parent
9de8abeb
Changes
3
Show whitespace changes
Inline
Side-by-side
Showing
3 changed files
with
413 additions
and
120 deletions
+413
-120
web/core/llm.py
web/core/llm.py
+248
-26
web/core/step_impl/step2_merge_redundancy.py
web/core/step_impl/step2_merge_redundancy.py
+90
-12
web/core/step_impl/step5_missing_comments.py
web/core/step_impl/step5_missing_comments.py
+75
-82
No files found.
web/core/llm.py
View file @
8ac75cef
...
...
@@ -219,12 +219,12 @@ class LLMClient:
# ── 核心调用 ──
def
complete
(
self
,
prompt
:
str
,
system
:
str
=
""
,
json_mode
:
bool
=
False
)
->
str
:
"""调用 LLM,返回纯文本。失败抛 LLMUnavailable。"""
"""调用 LLM,返回纯文本。失败抛 LLMUnavailable
(携带可读的错误信息)
。"""
if
not
self
.
available
:
raise
LLMUnavailable
(
"LLM 客户端未配置 API Key"
)
try
:
if
self
.
cfg
.
provider
==
"anthropic"
:
if
self
.
cfg
.
provider
in
(
"anthropic"
,
"minimax"
)
:
kwargs
:
dict
[
str
,
Any
]
=
{
"model"
:
self
.
cfg
.
model
,
"max_tokens"
:
self
.
cfg
.
max_tokens
,
...
...
@@ -248,9 +248,20 @@ class LLMClient:
kwargs
[
"response_format"
]
=
{
"type"
:
"json_object"
}
resp
=
self
.
_provider
.
chat
.
completions
.
create
(
**
kwargs
)
return
resp
.
choices
[
0
].
message
.
content
# 未支持的 provider——按理 _init_provider 已拦过,这里再防一次
raise
LLMUnavailable
(
f"complete() 未实现 provider=
{
self
.
cfg
.
provider
!
r
}
的调用分支"
)
except
LLMUnavailable
:
raise
except
Exception
as
e
:
logger
.
warning
(
f"LLM 调用失败:
{
e
}
"
)
raise
LLMUnavailable
(
str
(
e
))
detail
=
_format_api_error
(
e
,
self
.
cfg
.
provider
)
logger
.
warning
(
f"LLM 调用失败 (provider=
{
self
.
cfg
.
provider
}
, "
f"model=
{
self
.
cfg
.
effective_model
}
):
{
detail
}
"
)
raise
LLMUnavailable
(
detail
)
# ── JSON 模式(带解析兜底) ──
def
complete_json
(
self
,
prompt
:
str
,
system
:
str
=
""
)
->
dict
|
list
|
None
:
...
...
@@ -261,37 +272,141 @@ class LLMClient:
text
=
self
.
complete
(
prompt
,
system
=
json_system
,
json_mode
=
True
)
return
_safe_parse_json
(
text
)
# ── 业务方法:字段注释推测 ──
def
predict_field_comment
(
# ── 业务方法:字段注释推测
(批处理,required by Step 5)
──
def
predict_field_comment
s_batch
(
self
,
table_name
:
str
,
table_comment
:
str
,
column_name
:
str
,
data_type
:
str
,
)
->
dict
|
None
:
"""推测字段语义,返回 {"comment": str, "confidence": "high|medium|low"} 或 None"""
prompt
=
f"""你是数据库治理专家。请根据下面的信息,推测一个字段的中文注释。
fields
:
list
[
dict
],
)
->
list
[
dict
|
None
]:
"""批量推测字段语义,返回与输入等长的 list,None 表示该项失败。
表名:
{
table_name
}
表注释:
{
table_comment
or
'(无)'
}
字段名:
{
column_name
}
数据类型:
{
data_type
}
Input fields: [{"column_name": str, "data_type": str}, ...]
Output: [{"column_name": str, "comment": str,
"confidence": "high|medium|low"}, ...] 与输入同序
"""
if
not
fields
:
return
[]
if
not
self
.
available
:
raise
LLMUnavailable
(
"LLM 客户端未配置 API Key"
)
fields_json
=
"
\
n
"
.
join
(
f'
{
i
+
1
}
.
{
f
[
"column_name"
]
}
(
{
f
.
get
(
"data_type"
,
""
)
}
)'
for
i
,
f
in
enumerate
(
fields
)
)
prompt
=
f"""你是数据库治理专家。表名 `
{
table_name
}
`(
{
table_comment
or
'无注释'
}
)下有以下
{
len
(
fields
)
}
个无注释字段,请逐个推测中文注释。
字段清单:
{
fields_json
}
要求:
1. 给出 5-20 字的中文注释
2.
如字段名是拼音首字母缩写(如 xzqhbm),
翻译为中文
3.
如字段名是英文组合(如 create_time),
翻译为中文
1.
每条
给出 5-20 字的中文注释
2.
拼音首字母缩写(如 xzqhbm)
翻译为中文
3.
英文组合(如 create_time)
翻译为中文
4. 给出置信度(high=非常确定;medium=较确定;low=猜测)
5. 严格保持输入顺序,返回数组长度 ==
{
len
(
fields
)
}
严格返回 JSON 格式:{{"comment": "...", "confidence": "high|medium|low"}}
严格返回 JSON 数组(不要用对象包裹):
[
{{"column_name": "<原字段名>", "comment": "<中文注释>", "confidence": "high|medium|low"}},
...
]
"""
result
=
self
.
complete_json
(
prompt
)
if
isinstance
(
result
,
dict
)
and
"comment"
in
result
:
return
{
"comment"
:
str
(
result
[
"comment"
]).
strip
(),
"confidence"
:
result
.
get
(
"confidence"
,
"low"
),
}
return
None
if
not
isinstance
(
result
,
list
):
logger
.
warning
(
f"批量注释推测返回非数组:
{
type
(
result
).
__name__
}
"
)
return
[
None
]
*
len
(
fields
)
# 按 column_name 对齐回输入顺序(容错:LLM 可能打乱顺序)
by_name
=
{
r
.
get
(
"column_name"
):
r
for
r
in
result
if
isinstance
(
r
,
dict
)}
out
:
list
[
dict
|
None
]
=
[]
for
f
in
fields
:
r
=
by_name
.
get
(
f
[
"column_name"
])
if
r
and
r
.
get
(
"comment"
):
out
.
append
({
"column_name"
:
f
[
"column_name"
],
"comment"
:
str
(
r
[
"comment"
]).
strip
(),
"confidence"
:
r
.
get
(
"confidence"
,
"low"
),
})
else
:
out
.
append
(
None
)
return
out
# ── 业务方法:冗余字段分类(批处理,required by Step 2) ──
def
classify_redundant_fields_batch
(
self
,
fields
:
list
[
dict
],
)
->
list
[
dict
|
None
]:
"""批量分类高频字段是否为真冗余,返回与输入等长的 list。
Input fields: [{
"field": str, # 字段名
"table_count": int, # 出现表数
"sample_tables": [str], # 抽样表名(最多 5 张)
"sample_types": [str], # 抽样的数据类型列表
}, ...]
Output: [{
"field": str,
"classification": "common_base|common_business|suspicious|true_redundancy",
"reasoning": str,
"recommendation": str,
}, ...] 与输入同序
"""
if
not
fields
:
return
[]
if
not
self
.
available
:
raise
LLMUnavailable
(
"LLM 客户端未配置 API Key"
)
fields_json
=
"
\
n
"
.
join
(
f'
{
i
+
1
}
.
{
f
[
"field"
]
}
(出现在
{
f
.
get
(
"table_count"
,
"?"
)
}
张表,'
f'类型
{
","
.
join
(
f
.
get
(
"sample_types"
,
[])[:
3
])
or
"?"
}
,'
f'抽样表
{
", "
.
join
(
f
.
get
(
"sample_tables"
,
[])[:
5
])
}
)'
for
i
,
f
in
enumerate
(
fields
)
)
prompt
=
f"""数据库治理专家:以下
{
len
(
fields
)
}
个字段在数据库中出现频次较高,请判断它们的"高频出现"是否合理。
字段清单:
{
fields_json
}
分类维度(互斥):
- "common_base" :通用基础字段,所有表都该有(如 id / create_time / update_time / create_by)
- "common_business" :业务上合理共享(如 status / sort_order / type / code),可以保留
- "suspicious" :命名相同但含义可能不一致,需要核对每个表的具体定义
- "true_redundancy" :明确冗余——同名同义却被多表各自维护,应该抽公共字典或合并
要求:
1. 严格保持输入顺序,返回数组长度 ==
{
len
(
fields
)
}
2. reasoning 1-2 句说明判断依据
3. recommendation 给出整改建议(保留/合并/抽字典/核对)
严格返回 JSON 数组:
[
{{"field": "<原字段名>", "classification": "common_base|common_business|suspicious|true_redundancy",
"reasoning": "...", "recommendation": "..."}},
...
]
"""
result
=
self
.
complete_json
(
prompt
)
if
not
isinstance
(
result
,
list
):
logger
.
warning
(
f"批量冗余分类返回非数组:
{
type
(
result
).
__name__
}
"
)
return
[
None
]
*
len
(
fields
)
by_name
=
{
r
.
get
(
"field"
):
r
for
r
in
result
if
isinstance
(
r
,
dict
)}
out
:
list
[
dict
|
None
]
=
[]
for
f
in
fields
:
r
=
by_name
.
get
(
f
[
"field"
])
if
r
and
r
.
get
(
"classification"
)
in
(
"common_base"
,
"common_business"
,
"suspicious"
,
"true_redundancy"
):
out
.
append
({
"field"
:
f
[
"field"
],
"classification"
:
r
[
"classification"
],
"reasoning"
:
str
(
r
.
get
(
"reasoning"
,
""
)).
strip
(),
"recommendation"
:
str
(
r
.
get
(
"recommendation"
,
""
)).
strip
(),
})
else
:
out
.
append
(
None
)
return
out
# ── 业务方法:表合并建议 ──
def
suggest_table_merge
(
...
...
@@ -388,7 +503,114 @@ class LLMClient:
return
None
# ── JSON 解析兜底 ────────────────────────────────────────
# ── API 错误信息提取 ──────────────────────────────────────
def
_format_api_error
(
exc
:
Exception
,
provider
:
str
)
->
str
:
"""从 SDK 异常中提取可读的错误信息,供前端实时日志展示。
优先处理 Anthropic/OpenAI SDK 的结构化异常,
兜底处理网络/超时等通用异常。
"""
exc_type
=
type
(
exc
).
__name__
exc_msg
=
str
(
exc
)
# ── Anthropic SDK 异常(MiniMax 兼容接口也走这里) ──
try
:
from
anthropic
import
(
APIStatusError
,
RateLimitError
,
AuthenticationError
,
PermissionDeniedError
,
NotFoundError
,
APIConnectionError
,
APITimeoutError
,
)
if
isinstance
(
exc
,
RateLimitError
):
return
(
f"API Error: 请求被限流 (429) ·
{
_extract_body_message
(
exc
)
or
exc_msg
}
"
)
if
isinstance
(
exc
,
AuthenticationError
):
return
(
f"API Error: 认证失败 (401) · 请检查 API Key 是否正确或已过期"
)
if
isinstance
(
exc
,
PermissionDeniedError
):
return
(
f"API Error: 权限不足 (403) ·
{
_extract_body_message
(
exc
)
or
exc_msg
}
"
)
if
isinstance
(
exc
,
NotFoundError
):
return
(
f"API Error: 资源不存在 (404) ·
{
_extract_body_message
(
exc
)
or
exc_msg
}
"
)
if
isinstance
(
exc
,
APIStatusError
):
status
=
getattr
(
exc
,
"status_code"
,
"?"
)
body_msg
=
_extract_body_message
(
exc
)
detail
=
body_msg
or
exc_msg
# 对常见状态码给出中文提示
hint
=
{
429
:
"请升级 Token Plan 套餐或稍后重试"
,
500
:
"服务端内部错误,请稍后重试"
,
502
:
"网关错误,服务可能暂时不可用"
,
503
:
"服务暂不可用,请稍后重试"
,
}.
get
(
status
,
""
)
if
hint
:
return
f"API Error: 请求被拒绝 (
{
status
}
) ·
{
detail
}
(
{
hint
}
)"
return
f"API Error: 请求失败 (
{
status
}
) ·
{
detail
}
"
if
isinstance
(
exc
,
APIConnectionError
):
return
f"API Error: 网络连接失败 ·
{
exc_msg
}
"
if
isinstance
(
exc
,
APITimeoutError
):
return
f"API Error: 请求超时 ·
{
exc_msg
}
"
except
ImportError
:
pass
# ── OpenAI SDK 异常 ──
try
:
from
openai
import
(
APIStatusError
as
OAIStatusError
,
RateLimitError
as
OAIRateLimitError
,
AuthenticationError
as
OAIAuthError
,
APIConnectionError
as
OAIConnectionError
,
APITimeoutError
as
OAITimeoutError
,
)
if
isinstance
(
exc
,
OAIRateLimitError
):
return
(
f"API Error: 请求被限流 (429) ·
{
exc_msg
}
"
)
if
isinstance
(
exc
,
OAIAuthError
):
return
(
f"API Error: 认证失败 (401) · 请检查 API Key 是否正确或已过期"
)
if
isinstance
(
exc
,
OAIStatusError
):
status
=
getattr
(
exc
,
"status_code"
,
"?"
)
return
f"API Error: 请求失败 (
{
status
}
) ·
{
exc_msg
}
"
if
isinstance
(
exc
,
OAIConnectionError
):
return
f"API Error: 网络连接失败 ·
{
exc_msg
}
"
if
isinstance
(
exc
,
OAITimeoutError
):
return
f"API Error: 请求超时 ·
{
exc_msg
}
"
except
ImportError
:
pass
# ── 通用网络/超时异常 ──
import
builtins
if
isinstance
(
exc
,
builtins
.
ConnectionError
):
return
f"API Error: 网络连接失败 ·
{
exc_msg
}
"
if
isinstance
(
exc
,
builtins
.
TimeoutError
):
return
f"API Error: 请求超时 ·
{
exc_msg
}
"
# ── 兜底 ──
return
f"API Error:
{
exc_type
}
·
{
exc_msg
}
"
def
_extract_body_message
(
exc
:
Exception
)
->
str
|
None
:
"""尝试从 Anthropic APIStatusError 的 body 中提取错误详情。"""
body
=
getattr
(
exc
,
"body"
,
None
)
if
not
body
or
not
isinstance
(
body
,
dict
):
return
None
# body 结构: {"error": {"message": "..."}}
error
=
body
.
get
(
"error"
)
if
isinstance
(
error
,
dict
):
msg
=
error
.
get
(
"message"
)
if
msg
:
return
str
(
msg
)
return
None
def
_safe_parse_json
(
text
:
str
)
->
Any
:
"""从 LLM 返回中提取 JSON(容忍代码块包裹、尾部多余文字)"""
if
not
text
:
...
...
web/core/step_impl/step2_merge_redundancy.py
View file @
8ac75cef
...
...
@@ -61,8 +61,8 @@ def run_step2(dict_data: dict, llm: LLMClient | None = None,
# 3. 高频字段
if
log
:
log
(
"INFO"
,
"[3/3] 计算高频字段(出现 ≥10 张表)..."
,
step
=
"2"
)
redundancy
=
_find_redundancy
(
by_table
)
log
(
"INFO"
,
"[3/3] 计算高频字段(出现 ≥10 张表
,LLM 分类
)..."
,
step
=
"2"
)
redundancy
=
_find_redundancy
(
by_table
,
llm
,
log
)
if
log
:
log
(
"INFO"
,
f" · 高频字段
{
len
(
redundancy
)
}
个"
,
step
=
"2"
)
...
...
@@ -195,19 +195,97 @@ def _find_replacement(table: str, all_tables) -> str | None:
return
None
def
_find_redundancy
(
by_table
:
dict
)
->
list
[
dict
]:
def
_find_redundancy
(
by_table
:
dict
,
llm
,
log
:
Callable
|
None
)
->
list
[
dict
]:
"""高频字段分析:Counter 预筛 → LLM 分类(必跑)。
Counter 找出出现 ≥10 张表的字段(纯规则、毫秒级)。
然后对每个候选调 LLM 分类:
- common_base 通用基础字段,可保留
- common_business 业务上合理共享,可保留
- suspicious 命名相同但含义可能不一致,需核对
- true_redundancy 真冗余,建议合并 / 抽字典
LLM 必跑:若调用失败(网络/解析)会让该字段标记 llm_failed,不影响其他字段;
若整批不可用由上层(orchestrator)视为关键失败。
"""
# 1. Counter 预筛
counter
:
Counter
=
Counter
()
for
cols
in
by_table
.
values
():
type_by_field
:
dict
[
str
,
set
[
str
]]
=
{}
tables_by_field
:
dict
[
str
,
set
[
str
]]
=
{}
for
tname
,
cols
in
by_table
.
items
():
for
c
in
cols
:
counter
[
c
[
"column_name"
]]
+=
1
logger
.
debug
(
f"字段频次统计: 共
{
len
(
counter
)
}
个不同字段名"
)
return
[
fname
=
c
[
"column_name"
]
counter
[
fname
]
+=
1
type_by_field
.
setdefault
(
fname
,
set
()).
add
(
c
.
get
(
"data_type"
,
""
)
or
""
)
tables_by_field
.
setdefault
(
fname
,
set
()).
add
(
tname
)
candidates
=
[(
f
,
n
)
for
f
,
n
in
counter
.
most_common
()
if
n
>=
10
and
f
not
in
COMMON_FIELDS
]
logger
.
debug
(
f"字段频次统计: 共
{
len
(
counter
)
}
个不同字段名, "
f"候选高频字段
{
len
(
candidates
)
}
个"
)
if
not
candidates
:
return
[]
# 2. 构造 LLM 输入
llm_inputs
=
[
{
"field"
:
f
,
"table_count"
:
n
,
"
risk"
:
"高频出现,需评估是否为业务必要字段"
if
n
>=
30
else
"中频出现"
,
"s
uggestion"
:
"建议评审是否需要统一到公共字典表"
if
n
>=
30
else
"建议评审"
,
"
sample_tables"
:
sorted
(
tables_by_field
[
f
])[:
5
]
,
"s
ample_types"
:
sorted
(
type_by_field
[
f
])[:
3
]
,
}
for
f
,
n
in
counter
.
most_common
()
if
n
>=
10
and
f
not
in
COMMON_FIELDS
for
f
,
n
in
candidates
]
BATCH
=
15
annotated
:
list
[
dict
|
None
]
=
[]
total_batches
=
(
len
(
llm_inputs
)
+
BATCH
-
1
)
//
BATCH
for
batch_idx
in
range
(
0
,
len
(
llm_inputs
),
BATCH
):
batch
=
llm_inputs
[
batch_idx
:
batch_idx
+
BATCH
]
idx
=
batch_idx
//
BATCH
+
1
if
log
:
log
(
"INFO"
,
f" · LLM 分类批次 [
{
idx
}
/
{
total_batches
}
] "
f"(
{
len
(
batch
)
}
个字段)"
,
step
=
"2"
)
try
:
results
=
llm
.
classify_redundant_fields_batch
(
batch
)
except
Exception
as
e
:
if
log
:
log
(
"ERROR"
,
f" · LLM 分类批次 [
{
idx
}
/
{
total_batches
}
] 整体失败:
{
e
}
(任务将终止)"
,
step
=
"2"
)
logger
.
exception
(
"Step 2 LLM 分类批次失败"
)
raise
# required step → 抛给上层
annotated
.
extend
(
results
)
# 3. 汇总:每个候选都给出最终记录(LLM 成功的带 classification / reasoning / recommendation;
# LLM 失败的标 llm_failed,仍保留 field + table_count 便于定位)
out
:
list
[
dict
]
=
[]
for
cand
,
ann
in
zip
(
candidates
,
annotated
):
f
,
n
=
cand
if
ann
:
out
.
append
({
"field"
:
f
,
"table_count"
:
n
,
"classification"
:
ann
[
"classification"
],
"reasoning"
:
ann
[
"reasoning"
],
"recommendation"
:
ann
[
"recommendation"
],
"source"
:
"llm"
,
})
else
:
out
.
append
({
"field"
:
f
,
"table_count"
:
n
,
"classification"
:
"unknown"
,
"reasoning"
:
"LLM 解析失败"
,
"recommendation"
:
"需人工核对"
,
"source"
:
"llm_failed"
,
})
if
log
:
cls_count
=
{}
for
r
in
out
:
cls_count
[
r
[
"classification"
]]
=
cls_count
.
get
(
r
[
"classification"
],
0
)
+
1
summary
=
", "
.
join
(
f"
{
k
}
=
{
v
}
"
for
k
,
v
in
cls_count
.
items
())
if
log
:
log
(
"INFO"
,
f" · LLM 分类结果:
{
summary
}
"
,
step
=
"2"
)
return
out
\ No newline at end of file
web/core/step_impl/step5_missing_comments.py
View file @
8ac75cef
"""Step 5: 缺失注释字段检查 + LLM 推测
"""Step 5: 缺失注释字段检查 + LLM 推测
(必跑 LLM)
优先用 LLM 推测无注释字段的语义(替换原硬编码 COMMENT_HINTS);
LLM 不可用时降级为本地规则。
字段注释的语义判断本质上需要 LLM:
- 拼音首字母缩写(如 xzqhbm)、业务缩写、英文组合 → 必须由 LLM 翻译
- 硬编码字典兜底已删除(缺 LLM = 任务失败,无需 fallback)
产出:
- summary: 总数 +
推测
命中率
- summary: 总数 +
LLM
命中率
- by_table: 每表缺失注释数量
- predicted_comments:
推测出的注释(含置信度
)
- unpredict
able_sample: 未推测到的样本
- predicted_comments:
LLM 推测出的注释(含置信度、来源 = "llm"
)
- unpredict
ed_sample: LLM 单条失败的样本(标 llm_failed)
"""
from
__future__
import
annotations
...
...
@@ -21,31 +22,6 @@ from ..llm import LLMClient
logger
=
logging
.
getLogger
(
__name__
)
# 兜底规则(LLM 不可用时使用)
FALLBACK_HINTS
=
{
"id"
:
"主键ID"
,
"create_time"
:
"创建时间"
,
"update_time"
:
"更新时间"
,
"create_by"
:
"创建人"
,
"update_by"
:
"更新人"
,
"remark"
:
"备注"
,
"del_flag"
:
"删除标记"
,
"tenant_id"
:
"租户ID"
,
"dept_id"
:
"部门ID"
,
"project_id"
:
"项目ID"
,
"project_name"
:
"项目名称"
,
"project_code"
:
"项目编号"
,
"site_id"
:
"工地ID"
,
"site_name"
:
"工地名称"
,
"status"
:
"状态"
,
"sort_order"
:
"排序"
,
"start_time"
:
"开始时间"
,
"end_time"
:
"结束时间"
,
"type"
:
"类型"
,
"code"
:
"编码"
,
}
def
run_step5
(
dict_data
:
dict
,
llm
:
LLMClient
|
None
=
None
,
log
:
Callable
|
None
=
None
)
->
dict
:
columns
=
dict_data
.
get
(
"data_dictionary"
,
[])
...
...
@@ -57,19 +33,16 @@ def run_step5(dict_data: dict, llm: LLMClient | None = None,
if
log
:
log
(
"INFO"
,
f"待检查字段总数:
{
len
(
columns
)
}
"
,
step
=
"5"
)
# 1. 收集所有无注释字段,按 (table_name, table_comment) 分组
missing
=
[]
by_table
:
dict
[
str
,
int
]
=
defaultdict
(
int
)
predicted
=
[]
unpredictable
=
[]
grouped
:
dict
[
tuple
[
str
,
str
],
list
[
dict
]]
=
defaultdict
(
list
)
# 先用规则快速批匹配
for
r
in
columns
:
comment
=
(
r
.
get
(
"column_comment"
)
or
""
).
strip
()
if
comment
:
continue
by_table
[
r
[
"table_name"
]]
+=
1
fallback
=
FALLBACK_HINTS
.
get
(
r
[
"column_name"
])
entry
=
{
"table_name"
:
r
[
"table_name"
],
"table_comment"
:
r
.
get
(
"table_comment"
,
""
),
...
...
@@ -81,73 +54,77 @@ def run_step5(dict_data: dict, llm: LLMClient | None = None,
"confidence"
:
"low"
,
"reason"
:
""
,
}
if
fallback
:
entry
[
"predicted"
]
=
fallback
entry
[
"confidence"
]
=
"high"
entry
[
"reason"
]
=
"字段名匹配内置规则"
predicted
.
append
(
entry
)
else
:
unpredictable
.
append
(
entry
)
missing
.
append
(
entry
)
grouped
[(
r
[
"table_name"
],
r
.
get
(
"table_comment"
,
""
))].
append
(
entry
)
if
log
:
log
(
"INFO"
,
f" · 缺失注释字段:
{
len
(
missing
)
}
个 (覆盖
{
len
(
by_table
)
}
张表)"
,
step
=
"5"
)
log
(
"INFO"
,
f" · 内置规则命中:
{
len
(
predicted
)
}
, 需 LLM 推测:
{
len
(
unpredictable
)
}
"
,
step
=
"5"
)
# 用 LLM 处理 unpredictable(如果可用)
if
llm
and
llm
.
available
and
unpredictable
:
# 2. 按表分批调用 LLM(必跑;单批失败会让任务终止)
if
not
missing
:
if
log
:
log
(
"INFO"
,
" · 无缺失注释字段,跳过 LLM"
,
step
=
"5"
)
return
_empty_result
(
by_table
)
BATCH
=
15
predicted
=
[]
unpredicted
=
[]
llm_called
=
0
for
(
tname
,
tcomment
),
fields
in
grouped
.
items
():
# 把同表的字段切片成 BATCH 大小
chunks
=
[
fields
[
i
:
i
+
BATCH
]
for
i
in
range
(
0
,
len
(
fields
),
BATCH
)]
for
chunk_idx
,
chunk
in
enumerate
(
chunks
,
1
):
llm_called
+=
1
if
log
:
log
(
"INFO"
,
f"[LLM] 调用 LLM 推测
{
len
(
unpredictable
)
}
个无规则命中的字段注释"
,
step
=
"5"
)
llm_predicted
=
[]
still_unknown
=
[]
for
idx
,
entry
in
enumerate
(
unpredictable
,
1
):
log
(
"INFO"
,
f" · LLM 推测 [
{
llm_called
}
]
{
tname
}
"
f"(
{
chunk_idx
}
/
{
len
(
chunks
)
}
批,
{
len
(
chunk
)
}
个字段)"
,
step
=
"5"
)
try
:
r
=
llm
.
predict_field_comment
(
table_name
=
entry
[
"table_name"
]
,
table_comment
=
entry
[
"table_comment"
]
,
column_name
=
entry
[
"column_name"
],
data_type
=
entry
[
"data_type"
],
r
esults
=
llm
.
predict_field_comments_batch
(
table_name
=
tname
,
table_comment
=
tcomment
,
fields
=
[{
"column_name"
:
e
[
"column_name"
],
"data_type"
:
e
[
"data_type"
]}
for
e
in
chunk
],
)
except
Exception
as
e
:
# 单批失败 → 让任务终止(required step)
if
log
:
log
(
"ERROR"
,
f" · LLM 推测失败(
{
tname
}
第
{
chunk_idx
}
批):
{
e
}
(任务将终止)"
,
step
=
"5"
)
logger
.
exception
(
"Step 5 LLM 推测批次失败"
)
raise
# 把 LLM 结果写回 entry
for
entry
,
r
in
zip
(
chunk
,
results
):
if
r
and
r
.
get
(
"comment"
):
entry
[
"predicted"
]
=
r
[
"comment"
]
entry
[
"confidence"
]
=
r
.
get
(
"confidence"
,
"low"
)
entry
[
"reason"
]
=
"LLM 推测"
llm_predicted
.
append
(
entry
)
# 从 predicted 列表的视角也算推测成功
entry
[
"source"
]
=
"llm"
predicted
.
append
(
entry
)
if
log
and
idx
%
5
==
0
:
log
(
"DEBUG"
,
f" · LLM 推测进度
{
idx
}
/
{
len
(
unpredictable
)
}
"
f"(已成功
{
len
(
llm_predicted
)
}
)"
,
step
=
"5"
)
else
:
still_unknown
.
append
(
entry
)
except
Exception
as
e
:
if
log
:
log
(
"WARN"
,
f"LLM 推测失败 (
{
entry
[
'table_name'
]
}
.
{
entry
[
'column_name'
]
}
):
{
e
}
"
,
step
=
"5"
)
still_unknown
.
append
(
entry
)
if
log
:
log
(
"INFO"
,
f"[LLM] 推测成功
{
len
(
llm_predicted
)
}
/
{
len
(
unpredictable
)
}
"
,
step
=
"5"
)
unpredictable
=
still_unknown
entry
[
"reason"
]
=
"LLM 解析失败"
entry
[
"source"
]
=
"llm_failed"
unpredicted
.
append
(
entry
)
# 按表聚合
#
3.
按表聚合
by_table_list
=
sorted
(
[{
"table_name"
:
t
,
"missing_count"
:
c
}
for
t
,
c
in
by_table
.
items
()],
key
=
lambda
x
:
x
[
"missing_count"
],
reverse
=
True
)[:
20
]
if
log
:
log
(
"INFO"
,
f" · LLM 命中
{
len
(
predicted
)
}
/
{
len
(
missing
)
}
, "
f"LLM 解析失败
{
len
(
unpredicted
)
}
"
,
step
=
"5"
)
log
(
"INFO"
,
f"缺失注释
{
len
(
missing
)
}
个字段,覆盖
{
len
(
by_table
)
}
张表;"
f"推测成功
{
len
(
predicted
)
}
(规则
{
sum
(
1
for
p
in
predicted
if
p
.
get
(
'reason'
)
==
'字段名匹配内置规则'
)
}
, "
f"LLM
{
sum
(
1
for
p
in
predicted
if
p
.
get
(
'reason'
)
==
'LLM 推测'
)
}
)"
,
f"LLM 推测成功
{
len
(
predicted
)
}
"
,
step
=
"5"
)
return
{
...
...
@@ -155,11 +132,27 @@ def run_step5(dict_data: dict, llm: LLMClient | None = None,
"total_missing_comments"
:
len
(
missing
),
"tables_affected"
:
len
(
by_table
),
"predicted_count"
:
len
(
predicted
),
"predicted_by_
rules"
:
sum
(
1
for
p
in
predicted
if
p
.
get
(
"reason"
)
==
"字段名匹配内置规则"
),
"
predicted_by_llm"
:
sum
(
1
for
p
in
predicted
if
p
.
get
(
"reason"
)
==
"LLM 推测"
),
"
unpredicted_count"
:
len
(
unpredictable
)
,
"predicted_by_
llm"
:
len
(
predicted
),
"
unpredicted_count"
:
len
(
unpredicted
),
"
llm_calls"
:
llm_called
,
},
"by_table"
:
by_table_list
,
"predicted_comments"
:
predicted
[:
50
],
"unpredictable_sample"
:
unpredictable
[:
50
],
"unpredictable_sample"
:
unpredicted
[:
50
],
}
def
_empty_result
(
by_table
:
dict
)
->
dict
:
return
{
"summary"
:
{
"total_missing_comments"
:
0
,
"tables_affected"
:
len
(
by_table
),
"predicted_count"
:
0
,
"predicted_by_llm"
:
0
,
"unpredicted_count"
:
0
,
"llm_calls"
:
0
,
},
"by_table"
:
[],
"predicted_comments"
:
[],
"unpredictable_sample"
:
[],
}
\ No newline at end of file
Write
Preview
Markdown
is supported
0%
Try again
or
attach a new file
Attach a file
Cancel
You are about to add
0
people
to the discussion. Proceed with caution.
Finish editing this message first!
Cancel
Please
register
or
sign in
to comment