Commit d2cd3ad6 authored by Data Governance Dev's avatar Data Governance Dev

refactor(web): 移除 Step 3「数据验证」

Step 3 实际产出价值低且有副作用:
1. 「行政区划孤儿编码检查」只判断字典表 c_bri_xzqh 是否存在,
   没有真正跑 LEFT JOIN 找孤儿(代码注释里写了「可扩展」)
2. 「数据质量问题检查」只是占位,返回空列表
3. 「验证合并候选实际行数」有点用,但 result 已经直接覆盖 Step 2 的输出

最致命的是 _run_step3 返回 section_key='redundancy_fields' —— 这会
反向覆盖 Step 2 的 LLM 分类结果(classification / reasoning /
recommendation 全没了,只剩 Step 3 的简单 field + table_count 列表)。
也就是说 Step 2 的 LLM 工作被 Step 3 默默销毁了。移除 Step 3 同时也修了
这个覆盖 bug,Step 2 的分类结果能正确进入报告。

改动:

- 删除 web/core/step_impl/step3_verify.py
- 删除 web/sql/verify/ + web/sql/xzqh/ 整个目录(仅 Step 3 使用)
- orchestrator.py:
  - STEP_REGISTRY 移除 (3, 数据验证, _run_step3, ...)
  - STEP_DEPENDENCIES 移除 3: {1, 2}
  - step_name() 移除 3: '3_verify'
  - 删除 _run_step3 函数
  - 顶部 docstring 重写:6 步流程 + 移除 Step 3 的原因
- routes.py:list_steps() 不再返回 Step 3
- job_manager.py:all_steps 默认值改成从 STEP_REGISTRY 取
  (保持单一来源,避免硬编码漏改)
- frontend index.html:
  - form.steps 默认 [1,2,3,4,5,6,7] → [1,2,4,5,6,7]
  - 兜底过滤也加上 n !== 3(兼容旧本地缓存残留)

注:未触及 workflow/ 旧 CLI 目录,它走的是另一套独立实现。

行为:
- UI 上「执行步骤」区只有 6 个分析 Step
- /api/steps 返回 6 项
- 主流程跑完后无条件生成报告(Step 8 拆出来之前的约定不变)
- Step 2 LLM 分类结果不再被覆盖,能正确进冗余字段报告
parent fbedfe12
......@@ -74,9 +74,6 @@ async def list_steps():
StepInfo(num=2, title="表合并与冗余字段分析",
description="离线:找结构高度相似的表 + 高频字段(LLM 必须:合并 verdict 与冗余分类)",
requires_db=False, llm_mode="required", required=False),
StepInfo(num=3, title="数据验证",
description="连库验证:合并候选实际行数 + 行政区划孤儿 + 数据质量问题",
requires_db=True, llm_mode="none", required=False),
StepInfo(num=4, title="大范围空字段扫描",
description="单表全列扫,统计 NULL/空值比例(纯程序化检查,不依赖 LLM)",
requires_db=True, llm_mode="none", required=False),
......@@ -155,7 +152,7 @@ async def create_job(req: ConnectRequest):
job_id=job.job_id,
status=job.status,
created_at=job.created_at,
steps_planned=job.steps_planned or list(range(1, 8)),
steps_planned=job.steps_planned or [1, 2, 4, 5, 6, 7],
)
......
......@@ -187,8 +187,9 @@ class JobManager:
job.run_dir = run_dir
logger.info(f"[job {job.job_id}] 运行目录: {run_dir}")
# 收集要跑的步骤(7 个分析 Step;报告生成在主流程末尾无条件触发,不在此处)
all_steps = list(range(1, 8))
# 收集要跑的步骤(从 STEP_REGISTRY 取,保持单一来源;报告生成在主流程末尾无条件触发,不在此处)
from .orchestrator import STEP_REGISTRY
all_steps = [n for n, *_ in STEP_REGISTRY]
steps_planned = job.req.steps if job.req.steps else all_steps
job.steps_planned = steps_planned
logger.info(f"[job {job.job_id}] 计划步骤: {steps_planned}")
......
"""治理流程调度器
把 7 个分析 Step 串起来执行:
1. 数据字典 → 2. 合并/冗余(离线) → 3. 数据验证(连库)
把 6 个分析 Step 串起来执行:
1. 数据字典 → 2. 合并/冗余(离线)
→ 4. 空字段(连库) → 5. 缺注释(离线) → 6. 字段长度(离线)
→ 7. 国标校验(连库)
注:Step 3「数据验证」已移除——它实际产出是占位(xzqh 孤儿检查、数据质量均未实现),
而且会反向覆盖 Step 2 的 LLM 分类结果(返回 section_key='redundancy_fields'),
造成主步骤最关键的 LLM 输出被丢失。
报告生成不在 Step 列表内:主步骤全部跑完后无条件触发(一次性生成
Markdown + docx,供前端下载)。即使主步骤全部失败/跳过,也会尝试
基于已有的 sections 生成报告(可能只包含部分章节)。
Markdown + docx,供前端下载)。
特性:
- LLM 失败自动降级
......@@ -48,7 +51,6 @@ STEP_REGISTRY: list[tuple[int, str, Callable, bool, str, bool]] = [
# (num, title, fn, requires_db, llm_mode, required)
(1, "获取数据字典", "_run_step1", True, "none", True),
(2, "表合并与冗余字段分析", "_run_step2", False, "required", False),
(3, "数据验证", "_run_step3", True, "none", False),
(4, "大范围空字段扫描", "_run_step4", True, "none", False),
(5, "缺失注释检查 + 推测", "_run_step5", False, "required", False),
(6, "字段长度检查", "_run_step6", False, "none", False),
......@@ -62,7 +64,6 @@ STEP_REGISTRY: list[tuple[int, str, Callable, bool, str, bool]] = [
STEP_DEPENDENCIES: dict[int, set[int]] = {
1: set(), # 根基
2: {1},
3: {1, 2}, # 数据验证依赖 Step 2 的合并候选 / 冗余字段
4: set(), # 空字段扫描内部自取数据,可与 Step 1 并发
5: {1},
6: {1},
......@@ -391,7 +392,6 @@ def step_name(num: int) -> str:
return {
1: "1_data_dict",
2: "2_merge_redundancy",
3: "3_verify",
4: "4_empty_fields",
5: "5_missing_comments",
6: "6_length_check",
......@@ -415,17 +415,6 @@ def _run_step2(cfg, step_outputs, llm, findings_dir, log, cancel_event):
return {"section_key": "merge_candidates", "data": data}
def _run_step3(cfg, step_outputs, llm, findings_dir, log, cancel_event):
"""Step 3: 数据验证(连库)"""
from .step_impl.step3_verify import run_step3
s2 = step_outputs.get("2_merge_redundancy", {})
data = run_step3(cfg, dict_data=step_outputs.get("1_data_dict", {}),
merge_candidates=s2.get("merge_candidates", []),
redundancy_fields=s2.get("redundancy_fields", []),
log=log)
return {"section_key": "redundancy_fields", "data": data}
def _run_step4(cfg, step_outputs, llm, findings_dir, log, cancel_event):
"""Step 4: 空字段扫描(连库)—— 复用 Step 1 已拿到的数据字典"""
from .step_impl.step4_empty_fields import run_step4
......
"""Step 3: 数据验证(连库)
连库验证:
- merge_candidates: 对每组前 5 张表跑 SELECT COUNT(*)
- redundancy_fields: 取高频字段,看实际出现在哪些表
- xzqh_orphan_check: 行政区划编码孤儿(如果存在字典表 c_bri_xzqh)
- data_quality_issues: 数据质量问题清单
所有 SQL 走 web/sql/ 模板,通过 db_adapter 调用。
"""
from __future__ import annotations
import logging
from typing import Callable
from ..db_adapter import DBConfig, open_db, quote_ident
from web.sql.loader import get_sql_loader
logger = logging.getLogger(__name__)
def run_step3(cfg: DBConfig, dict_data: dict,
merge_candidates: list, redundancy_fields: list,
log: Callable | None = None) -> dict:
confirmed_merges = []
redundancy_verified = []
xzqh_check = {"orphan_total": 0, "per_table": []}
quality_issues = []
if log:
log("INFO", f"[1/4] 验证合并候选(top {min(5, len(merge_candidates))})...", step="3")
try:
with open_db(cfg) as db:
loader = get_sql_loader()
# 1. 验证 merge 候选(实际行数)
for idx, mc in enumerate(merge_candidates[:5], 1):
row_counts = []
total = 0
for t in mc.get("tables", []):
try:
sql = loader.render(
"verify/count_table_rows",
dialect=cfg.db_type,
table=t, # 模板内 ${table | quote} 自动加引号
)
n = db.fetch_scalar(sql) or 0
row_counts.append({"table": t, "rows": int(n)})
total += int(n)
except Exception as e:
row_counts.append({"table": t, "error": str(e)})
if log:
log("WARN", f" · 表 {t} 行数统计失败: {e}", step="3")
confirmed_merges.append({
**mc,
"row_counts": row_counts,
"total_rows": total,
})
if log:
log("INFO",
f" · {mc.get('id')} ({idx}/{min(5, len(merge_candidates))}): "
f"{len(mc.get('tables', []))} 张表, 共 {total} 行",
step="3")
# 2. 验证冗余字段(出现表数 + 实际行数)
if log:
log("INFO", f"[2/4] 验证冗余字段(top {min(5, len(redundancy_fields))})...", step="3")
for rf in redundancy_fields[:5]:
field_name = rf.get("field")
tables = _find_field_in_tables(dict_data.get("data_dictionary", []), field_name)
redundancy_verified.append({
"field": field_name,
"occurrences": len(tables),
"tables_with_field": tables,
})
if log:
log("INFO", f" · 字段 {field_name} 出现在 {len(tables)} 张表", step="3")
# 3. 行政区划孤儿编码检查
if log:
log("INFO", "[3/4] 行政区划孤儿编码检查...", step="3")
xzqh_check = _check_xzqh_orphans(cfg, db, log)
# 4. 数据质量问题
if log:
log("INFO", "[4/4] 数据质量问题检查(占位)...", step="3")
quality_issues = []
except Exception as e:
if log:
log("ERROR", f"Step 3 连库验证失败: {type(e).__name__}: {e}", step="3")
logger.exception("Step 3 详细异常")
if log:
log("INFO",
f"Step 3 完成: 验证合并 {len(confirmed_merges)} 组, "
f"冗余字段 {len(redundancy_verified)} 个, 行政区划孤儿 {xzqh_check.get('orphan_total', 0)} 条",
step="3")
return {
"confirmed_merge_candidates": confirmed_merges,
"redundancy_fields": redundancy_verified,
"xzqh_orphan_check": xzqh_check,
"data_quality_issues": quality_issues,
}
def _find_field_in_tables(columns: list, field_name: str) -> list[str]:
return sorted({r["table_name"] for r in columns if r["column_name"] == field_name})
def _check_xzqh_orphans(cfg: DBConfig, db, log: Callable | None) -> dict:
"""检查行政区划字段是否存在孤儿编码(不在字典表 c_bri_xzqh 中)"""
DICT_TABLE = "c_bri_xzqh"
result = {"dict_table_rows": 0, "orphan_total": 0, "per_table": []}
loader = get_sql_loader()
# 检查字典表是否存在 + 行数
try:
sql = loader.render(
"xzqh/check_dict_table_exists",
dialect=cfg.db_type,
dict_table=DICT_TABLE, # 模板内 ${dict_table | quote} 自动加引号
)
result["dict_table_rows"] = int(db.fetch_scalar(sql) or 0)
except Exception:
if log:
log("INFO", f"字典表 {DICT_TABLE} 不存在,跳过行政区划校验", step="3")
return result
if result["dict_table_rows"] == 0:
return result
if log:
log("INFO", f"字典表 {DICT_TABLE} 存在,共 {result['dict_table_rows']} 行", step="3")
# 这里可扩展:对含 xzqhbm 字段的前 5 张大表跑 LEFT JOIN 找孤儿
# 模板:xzqh/find_xzqh_orphans.sql
return result
\ No newline at end of file
-- ============================================================================
-- 统计单表的实际行数
-- 调用方:web/core/step_impl/step3_verify.py
-- 参数:${table} 表名(自动加引号:MySQL 反引号 / 达梦双引号)
-- ============================================================================
SELECT COUNT(*) AS n FROM ${table | quote}
\ No newline at end of file
-- ============================================================================
-- 检查行政区划字典表是否存在
-- 调用方:web/core/step_impl/step3_verify.py
-- 参数:${dict_table} 字典表名(自动加引号)
-- 返回:单行单列(行数;>0 表示表存在且非空)
-- ============================================================================
SELECT COUNT(*) FROM ${dict_table | quote}
\ No newline at end of file
-- ============================================================================
-- 找出业务表中存在但字典表中不存在的行政区划编码(孤儿)
-- 调用方:web/core/step_impl/step3_verify.py
-- 参数:
-- ${table} 业务表名(自动加引号)
-- ${xzqh_field} 行政区划字段名(自动加引号)
-- ${dict_table} 字典表名(自动加引号)
-- ${dict_code} 字典表编码字段名(自动加引号)
-- ${sample_limit} 抽样上限
-- ============================================================================
SELECT t.${xzqh_field | quote} AS orphan_code, COUNT(*) AS occurrences
FROM ${table | quote} t
LEFT JOIN ${dict_table | quote} d ON t.${xzqh_field | quote} = d.${dict_code | quote}
WHERE t.${xzqh_field | quote} IS NOT NULL
AND TRIM(t.${xzqh_field | quote}) <> ''
AND d.${dict_code | quote} IS NULL
GROUP BY t.${xzqh_field | quote}
ORDER BY occurrences DESC
LIMIT ${sample_limit}
\ No newline at end of file
......@@ -732,7 +732,7 @@
database: 'smart-build',
charset: 'utf8mb4',
connect_timeout: 10,
steps: [1, 2, 3, 4, 5, 6, 7],
steps: [1, 2, 4, 5, 6, 7],
});
const testing = ref(false);
......@@ -874,8 +874,8 @@
form.steps.push(s.num);
}
}
// 兜底:报告生成不再是 Step,不在 form.steps 里出现(兼容旧本地缓存里残留的 8)
form.steps = form.steps.filter(n => n !== 8);
// 兜底:报告生成不再是 Step,不在 form.steps 里出现(兼容旧本地缓存里残留的 8);Step 3 已移除
form.steps = form.steps.filter(n => n !== 8 && n !== 3);
try {
const r = await fetch('/api/jobs', {
method: 'POST',
......
Markdown is supported
0%
or
You are about to add 0 people to the discussion. Proceed with caution.
Finish editing this message first!
Please register or to comment