Commit 472a0c4a authored by Data Governance Dev's avatar Data Governance Dev

refactor(web): 重新编号 Step — 4-7 → 3-6,让步骤连续 1-6

移除 Step 3「数据验证」后,Step 号出现断层:1,2,4,5,6,7。重新编号让 Step 号
连续 1-6,对用户更直观。

改动:

orchestrator.py:
- STEP_REGISTRY:4→3 (空字段扫描), 5→4 (缺注释), 6→5 (字段长度), 7→6 (国标校验)
- STEP_DEPENDENCIES 同步重新编号
- step_name() 同步重新编号('3_empty_fields' / '4_missing_comments' / 等)
- _run_step4/5/6/7 重命名为 _run_step3/4/5/6(调用方 STEP_REGISTRY 已对,自动匹配)
- 顶部 docstring 重写:6 步流程 + 重新编号说明

step_impl/*.py:每个文件里的 step="X" 日志标签同步重新编号
(4→3, 5→4, 6→5, 7→6),同时把 docstring 和 message 文本里的 Step X 字样也改了。
文件本身保留旧文件名(如 step4_empty_fields.py)—— 内部实现细节,
不影响外部行为;如果未来想重命名可单独一次提交。

routes.py:list_steps() 的 StepInfo 也重新编号(4→3, 5→4, 6→5, 7→6)

frontend index.html:
- form.steps 默认 [1,2,4,5,6,7] → [1,2,3,4,5,6]
- 兜底过滤改成 n >= 1 && n <= 6(兼容旧缓存残留的 8/7/3)
- TAB_STEP_MAP:
  - redundancy tab:step 3 → 2(与 merge 同一来源)
  - empty: 4 → 3
  - missing: 5 → 4
  - length: 6 → 5
  - standards: 7 → 6

注:未触及 workflow/ 旧 CLI 目录、docs/DESIGN.md / README.md 文档,
那些是历史描述,单独一次文档 PR 一起改。

行为:UI 显示 Step 1-6 连续 6 个分析 Step,进度计算 / 报告生成
/ section 关联都不变。文件 / 函数命名差异仅是内部实现细节。
parent d2cd3ad6
......@@ -74,16 +74,16 @@ async def list_steps():
StepInfo(num=2, title="表合并与冗余字段分析",
description="离线:找结构高度相似的表 + 高频字段(LLM 必须:合并 verdict 与冗余分类)",
requires_db=False, llm_mode="required", required=False),
StepInfo(num=4, title="大范围空字段扫描",
StepInfo(num=3, title="大范围空字段扫描",
description="单表全列扫,统计 NULL/空值比例(纯程序化检查,不依赖 LLM)",
requires_db=True, llm_mode="none", required=False),
StepInfo(num=5, title="缺失注释字段 + 推测",
StepInfo(num=4, title="缺失注释字段 + 推测",
description="离线:找无注释字段,LLM 必须:批量推测语义",
requires_db=False, llm_mode="required", required=False),
StepInfo(num=6, title="字段长度检查",
StepInfo(num=5, title="字段长度检查",
description="离线:识别字段定义长度超出国家标准(纯规则匹配,不依赖 LLM)",
requires_db=False, llm_mode="none", required=False),
StepInfo(num=7, title="国家标准校验",
StepInfo(num=6, title="国家标准校验",
description="连库抽样:身份证/USCC/手机号/行政区划符合性",
requires_db=True, llm_mode="none", required=False),
])
......
......@@ -2,12 +2,13 @@
把 6 个分析 Step 串起来执行:
1. 数据字典 → 2. 合并/冗余(离线)
→ 4. 空字段(连库) → 5. 缺注释(离线) → 6. 字段长度(离线)
→ 7. 国标校验(连库)
→ 3. 空字段(连库) → 4. 缺注释(离线) → 5. 字段长度(离线)
→ 6. 国标校验(连库)
注:Step 3「数据验证」已移除——它实际产出是占位(xzqh 孤儿检查、数据质量均未实现),
而且会反向覆盖 Step 2 的 LLM 分类结果(返回 section_key='redundancy_fields'),
造成主步骤最关键的 LLM 输出被丢失。
历史:早期版本有 Step 3「数据验证」,但产出是占位(xzqh 孤儿检查、
数据质量均未实现),且会反向覆盖 Step 2 的 LLM 分类结果(返回
section_key='redundancy_fields')。已移除,并顺手重新编号
4-7 → 3-6 让 Step 号连续。
报告生成不在 Step 列表内:主步骤全部跑完后无条件触发(一次性生成
Markdown + docx,供前端下载)。
......@@ -41,9 +42,9 @@ logger = logging.getLogger(__name__)
# 顺序敏感:前面 Step 的产出可被后面 Step 复用
#
# llm_mode 取值:
# "none" —— 不用 LLM(Step 1、3、4、6、7)
# "none" —— 不用 LLM(Step 1、3、5、6)
# "optional" —— LLM 可用就用,失败降级到规则推理(当前未使用)
# "required" —— 必须 LLM;缺 LLM = 任务失败(Step 2、5)
# "required" —— 必须 LLM;缺 LLM = 任务失败(Step 2、4)
#
# required 标记:true 表示用户在 UI 上不能取消勾选(如 Step 1 为全部依赖的根基)
# 报告生成不再作为 Step 8 注册——它是流程结束后的固定动作,见 _generate_reports
......@@ -51,10 +52,10 @@ STEP_REGISTRY: list[tuple[int, str, Callable, bool, str, bool]] = [
# (num, title, fn, requires_db, llm_mode, required)
(1, "获取数据字典", "_run_step1", True, "none", True),
(2, "表合并与冗余字段分析", "_run_step2", False, "required", False),
(4, "大范围空字段扫描", "_run_step4", True, "none", False),
(5, "缺失注释检查 + 推测", "_run_step5", False, "required", False),
(6, "字段长度检查", "_run_step6", False, "none", False),
(7, "国家标准校验", "_run_step7", True, "none", False),
(3, "大范围空字段扫描", "_run_step3", True, "none", False),
(4, "缺失注释检查 + 推测", "_run_step4", False, "required", False),
(5, "字段长度检查", "_run_step5", False, "none", False),
(6, "国家标准校验", "_run_step6", True, "none", False),
]
# ── Step 依赖图 ───────────────────────────────────────────
......@@ -64,10 +65,10 @@ STEP_REGISTRY: list[tuple[int, str, Callable, bool, str, bool]] = [
STEP_DEPENDENCIES: dict[int, set[int]] = {
1: set(), # 根基
2: {1},
4: set(), # 空字段扫描内部自取数据,可与 Step 1 并发
3: set(), # 空字段扫描内部自取数据,可与 Step 1 并发
4: {1},
5: {1},
6: {1},
7: {1},
}
# ── 关键 Step 集合 ────────────────────────────────────────
......@@ -392,10 +393,10 @@ def step_name(num: int) -> str:
return {
1: "1_data_dict",
2: "2_merge_redundancy",
4: "4_empty_fields",
5: "5_missing_comments",
6: "6_length_check",
7: "7_standards",
3: "3_empty_fields",
4: "4_missing_comments",
5: "5_length_check",
6: "6_standards",
}[num]
......@@ -415,31 +416,31 @@ def _run_step2(cfg, step_outputs, llm, findings_dir, log, cancel_event):
return {"section_key": "merge_candidates", "data": data}
def _run_step4(cfg, step_outputs, llm, findings_dir, log, cancel_event):
"""Step 4: 空字段扫描(连库)—— 复用 Step 1 已拿到的数据字典"""
def _run_step3(cfg, step_outputs, llm, findings_dir, log, cancel_event):
"""Step 3: 空字段扫描(连库)—— 复用 Step 1 已拿到的数据字典"""
from .step_impl.step4_empty_fields import run_step4
dict_data = step_outputs.get("1_data_dict") or {}
data = run_step4(cfg, dict_data=dict_data, log=log)
return {"section_key": "empty_fields", "data": data}
def _run_step5(cfg, step_outputs, llm, findings_dir, log, cancel_event):
"""Step 5: 缺注释 + LLM 推测"""
def _run_step4(cfg, step_outputs, llm, findings_dir, log, cancel_event):
"""Step 4: 缺注释 + LLM 推测"""
from .step_impl.step5_missing_comments import run_step5
data = run_step5(dict_data=step_outputs.get("1_data_dict", {}),
llm=llm, log=log)
return {"section_key": "missing_comments", "data": data}
def _run_step6(cfg, step_outputs, llm, findings_dir, log, cancel_event):
"""Step 6: 字段长度检查(离线)"""
def _run_step5(cfg, step_outputs, llm, findings_dir, log, cancel_event):
"""Step 5: 字段长度检查(离线)"""
from .step_impl.step6_length_check import run_step6
data = run_step6(dict_data=step_outputs.get("1_data_dict", {}), log=log)
return {"section_key": "length_issues", "data": data}
def _run_step7(cfg, step_outputs, llm, findings_dir, log, cancel_event):
"""Step 7: 国标校验(连库)"""
def _run_step6(cfg, step_outputs, llm, findings_dir, log, cancel_event):
"""Step 6: 国标校验(连库)"""
from .step_impl.step7_standards import run_step7
data = run_step7(cfg, dict_data=step_outputs.get("1_data_dict", {}), log=log)
return {"section_key": "standard_violations", "data": data}
......
"""Step 4: 大范围空字段扫描(纯程序化检查)
"""Step 3: 大范围空字段扫描(纯程序化检查)
对每张表跑动态 SQL 统计:
- COUNT(*) 总行数
......@@ -39,23 +39,23 @@ def run_step4(cfg: DBConfig, log: Callable | None = None,
"""
if log:
log("INFO",
f"Step 4 开始扫描空字段 (high≥{high_threshold:.0%}, mid≥{mid_threshold:.0%}, min_rows={min_rows_to_check})",
step="4")
f"Step 3 开始扫描空字段 (high≥{high_threshold:.0%}, mid≥{mid_threshold:.0%}, min_rows={min_rows_to_check})",
step="3")
if dict_data and dict_data.get("data_dictionary"):
# 复用 Step 1 产出(Wave 调度器已保证 Step 1 先完成)
columns = dict_data["data_dictionary"]
table_summary = dict_data.get("table_summary", [])
if log:
log("INFO", " · 复用 Step 1 数据字典", step="4")
log("INFO", " · 复用 Step 1 数据字典", step="3")
else:
# 兜底:没拿到 Step 1 产出时回退自取(独立跑 Step 4 时用得到)
# 兜底:没拿到 Step 1 产出时回退自取(独立跑 Step 3 时用得到)
from .step1_data_dict import run_step1
dict_data = run_step1(cfg, log=lambda lvl, msg, **_kw: log(lvl, msg, step="4") if log else None)
dict_data = run_step1(cfg, log=lambda lvl, msg, **_kw: log(lvl, msg, step="3") if log else None)
columns = dict_data["data_dictionary"]
table_summary = dict_data.get("table_summary", [])
if log:
log("INFO", " · 独立拉取数据字典(未复用 Step 1)", step="4")
log("INFO", " · 独立拉取数据字典(未复用 Step 1)", step="3")
by_table: dict[str, list[dict]] = defaultdict(list)
for row in columns:
......@@ -88,15 +88,15 @@ def run_step4(cfg: DBConfig, log: Callable | None = None,
table_names = sorted(by_table.keys())
total = len(table_names)
if log:
log("INFO", f"将扫描 {total} 张表(已过滤 TEXT_TYPES)", step="4")
log("INFO", f"将扫描 {total} 张表(已过滤 TEXT_TYPES)", step="3")
try:
with open_db(cfg) as db:
for idx, table in enumerate(table_names, 1):
if log and idx % 10 == 0:
log("INFO", f" 进度 {idx}/{total} - 当前表: {table}", step="4")
log("INFO", f" 进度 {idx}/{total} - 当前表: {table}", step="3")
elif log and idx == 1:
log("INFO", f" 进度 1/{total} - 当前表: {table}", step="4")
log("INFO", f" 进度 1/{total} - 当前表: {table}", step="3")
# 拿实际行数(避免 TABLE_ROWS 估算不准)
try:
......@@ -109,7 +109,7 @@ def run_step4(cfg: DBConfig, log: Callable | None = None,
except Exception as e:
skipped.append({"table": table, "reason": f"统计行数失败: {e}"})
if log:
log("WARN", f" · {table} 行数统计失败: {e}", step="4")
log("WARN", f" · {table} 行数统计失败: {e}", step="3")
continue
if row_count < min_rows_to_check:
......@@ -134,7 +134,7 @@ def run_step4(cfg: DBConfig, log: Callable | None = None,
except Exception as e:
skipped.append({"table": table, "reason": f"扫描失败: {e}"})
if log:
log("WARN", f" · {table} 空字段扫描失败: {e}", step="4")
log("WARN", f" · {table} 空字段扫描失败: {e}", step="3")
continue
if not row:
......@@ -189,8 +189,8 @@ def run_step4(cfg: DBConfig, log: Callable | None = None,
logger.debug(f"表 {table}: 高空 {len(high_in_table)}, 中空 {len(mid_in_table)}")
except Exception as e:
if log:
log("ERROR", f"Step 4 数据库扫描失败: {type(e).__name__}: {e}", step="4")
logger.exception("Step 4 详细异常")
log("ERROR", f"Step 3 数据库扫描失败: {type(e).__name__}: {e}", step="3")
logger.exception("Step 3 详细异常")
top_tables.sort(key=lambda x: x["high_empty_field_count"], reverse=True)
......@@ -210,7 +210,7 @@ def run_step4(cfg: DBConfig, log: Callable | None = None,
log("INFO",
f"扫描完成: 高空 {high_count}, 中空 {mid_count}, 涉及 {len(tables_with_issues)} 张表, "
f"跳过 {len(skipped)} 张",
step="4")
step="3")
return {
"summary": summary,
......
"""Step 5: 缺失注释字段检查 + LLM 推测(必跑 LLM)
"""Step 4: 缺失注释字段检查 + LLM 推测(必跑 LLM)
字段注释的语义判断本质上需要 LLM:
- 拼音首字母缩写(如 xzqhbm)、业务缩写、英文组合 → 必须由 LLM 翻译
......@@ -27,11 +27,11 @@ def run_step5(dict_data: dict, llm: LLMClient | None = None,
columns = dict_data.get("data_dictionary", [])
if not columns:
if log:
log("WARN", "未获取到任何字段元数据,跳过 Step 5", step="5")
log("WARN", "未获取到任何字段元数据,跳过 Step 4", step="4")
return {"summary": {}, "by_table": [], "predicted_comments": [], "unpredictable_sample": []}
if log:
log("INFO", f"待检查字段总数: {len(columns)}", step="5")
log("INFO", f"待检查字段总数: {len(columns)}", step="4")
# 1. 收集所有无注释字段,按 (table_name, table_comment) 分组
missing = []
......@@ -60,12 +60,12 @@ def run_step5(dict_data: dict, llm: LLMClient | None = None,
if log:
log("INFO",
f" · 缺失注释字段: {len(missing)} 个 (覆盖 {len(by_table)} 张表)",
step="5")
step="4")
# 2. 按表分批调用 LLM(必跑;单批失败会让任务终止)
if not missing:
if log:
log("INFO", " · 无缺失注释字段,跳过 LLM", step="5")
log("INFO", " · 无缺失注释字段,跳过 LLM", step="4")
return _empty_result(by_table)
BATCH = 30
......@@ -81,7 +81,7 @@ def run_step5(dict_data: dict, llm: LLMClient | None = None,
log("INFO",
f" · LLM 推测 [{llm_called}] {tname} "
f"({chunk_idx}/{len(chunks)} 批, {len(chunk)} 个字段)",
step="5")
step="4")
try:
results = llm.predict_field_comments_batch(
table_name=tname,
......@@ -94,8 +94,8 @@ def run_step5(dict_data: dict, llm: LLMClient | None = None,
if log:
log("ERROR",
f" · LLM 推测失败({tname} 第 {chunk_idx} 批): {e}(任务将终止)",
step="5")
logger.exception("Step 5 LLM 推测批次失败")
step="4")
logger.exception("Step 4 LLM 推测批次失败")
raise
# 把 LLM 结果写回 entry
......@@ -121,11 +121,11 @@ def run_step5(dict_data: dict, llm: LLMClient | None = None,
log("INFO",
f" · LLM 命中 {len(predicted)}/{len(missing)}, "
f"LLM 解析失败 {len(unpredicted)}",
step="5")
step="4")
log("INFO",
f"缺失注释 {len(missing)} 个字段,覆盖 {len(by_table)} 张表;"
f"LLM 推测成功 {len(predicted)}",
step="5")
step="4")
return {
"summary": {
......
"""Step 6: 字段长度检查(纯规则匹配)
"""Step 5: 字段长度检查(纯规则匹配)
识别字段定义长度超出标准所需(如身份证号 18 位用 varchar(50) 存储)。
仅基于内置规则匹配,不依赖 LLM。
......@@ -156,13 +156,13 @@ def run_step6(dict_data: dict, log: Callable | None = None) -> dict:
columns = dict_data.get("data_dictionary", [])
if not columns:
if log:
log("WARN", "未获取到任何字段元数据,跳过 Step 6", step="6")
log("WARN", "未获取到任何字段元数据,跳过 Step 5", step="5")
return {"summary": {"total_issues": 0}, "issues": []}
if log:
log("INFO",
f"内置长度规则: {len(LENGTH_RULES)} 条, 待扫描字段: {len(columns)}",
step="6")
step="5")
issues = []
match_by_name = 0
......@@ -209,7 +209,7 @@ def run_step6(dict_data: dict, log: Callable | None = None) -> dict:
log("INFO",
f"规则匹配: 字段名 {match_by_name} 次 + 注释关键字 {match_by_comment} 次,"
f"发现异常 {len(issues)} 条",
step="6")
step="5")
return {
"summary": {
......
"""Step 7: 国家标准字段校验(连库)
"""Step 6: 国家标准字段校验(连库)
按字段名匹配 standards 插件,抽样校验实际数据:
- 身份证号(GB 11643)
......@@ -46,7 +46,7 @@ SAMPLE_SPECS: list[dict] = [
def run_step7(cfg: DBConfig, dict_data: dict, log: Callable | None = None) -> dict:
"""执行 Step 7:国标校验"""
"""执行 Step 6:国标校验"""
columns = dict_data.get("data_dictionary", [])
if not columns:
return {"summary": {}, "violations_by_standard": [], "violations": []}
......@@ -67,7 +67,7 @@ def run_step7(cfg: DBConfig, dict_data: dict, log: Callable | None = None) -> di
if not standards_by_field:
if log:
log("WARN", "未发现任何标准插件", step="7")
log("WARN", "未发现任何标准插件", step="6")
return {
"summary": {"standards_applied": 0, "fields_checked": 0, "violations": 0},
"violations_by_standard": [],
......@@ -79,7 +79,7 @@ def run_step7(cfg: DBConfig, dict_data: dict, log: Callable | None = None) -> di
f"已加载 {len(all_standards_info)} 个标准插件, "
f"匹配字段 {len(standards_by_field)} 个, "
f"最大每字段抽样 {SAMPLE_LIMIT} 行 / 每字段最多 {MAX_TABLES_PER_FIELD} 张表",
step="7")
step="6")
loader = get_sql_loader()
......@@ -90,7 +90,7 @@ def run_step7(cfg: DBConfig, dict_data: dict, log: Callable | None = None) -> di
if log:
log("DEBUG",
f" · [{std_idx}/{len(standards_by_field)}] 标准字段 {std_field} 未出现在任何表中, 跳过",
step="7")
step="6")
continue
tables_with_field = by_name[std_field][:MAX_TABLES_PER_FIELD]
std_id = std_instance.standard_id
......@@ -107,7 +107,7 @@ def run_step7(cfg: DBConfig, dict_data: dict, log: Callable | None = None) -> di
log("INFO",
f" · [{std_idx}/{len(standards_by_field)}] "
f"{std_id} ({std_field}) → {len(tables_with_field)} 张表",
step="7")
step="6")
for col_record in tables_with_field:
table = col_record["table_name"]
......@@ -139,7 +139,7 @@ def run_step7(cfg: DBConfig, dict_data: dict, log: Callable | None = None) -> di
if log:
log("WARN",
f" - {table}.{field_name} 抽样失败: {e}",
step="7")
step="6")
continue
sampled = len(rows)
......@@ -188,8 +188,8 @@ def run_step7(cfg: DBConfig, dict_data: dict, log: Callable | None = None) -> di
)
except Exception as e:
if log:
log("ERROR", f"Step 7 连库校验失败: {type(e).__name__}: {e}", step="7")
logger.exception("Step 7 详细异常")
log("ERROR", f"Step 6 连库校验失败: {type(e).__name__}: {e}", step="6")
logger.exception("Step 6 详细异常")
by_standard_list = []
for std_id, rec in by_standard.items():
......@@ -198,7 +198,7 @@ def run_step7(cfg: DBConfig, dict_data: dict, log: Callable | None = None) -> di
if log:
log("INFO",
f" · 标准 {rec['standard']} 抽样 {rec['total_sampled']} 条全部合规,跳过",
step="7")
step="6")
continue
violating_fields = [d for d in rec["fields"] if d.get("violations", 0) > 0]
by_standard_list.append({
......@@ -219,7 +219,7 @@ def run_step7(cfg: DBConfig, dict_data: dict, log: Callable | None = None) -> di
})
if log:
log("INFO", f"校验完成: {len(by_standard_list)} 个标准有违规, {total_fields_checked} 个字段, {total_violations} 条违规", step="7")
log("INFO", f"校验完成: {len(by_standard_list)} 个标准有违规, {total_fields_checked} 个字段, {total_violations} 条违规", step="6")
return {
"summary": {
......
......@@ -732,7 +732,7 @@
database: 'smart-build',
charset: 'utf8mb4',
connect_timeout: 10,
steps: [1, 2, 4, 5, 6, 7],
steps: [1, 2, 3, 4, 5, 6],
});
const testing = ref(false);
......@@ -773,11 +773,11 @@
// tab → (所属 step, 对应 sections 里的 key)
const TAB_STEP_MAP = {
merge: { step: 2, sectionKey: 'merge_candidates' },
redundancy: { step: 3, sectionKey: 'redundancy_fields' },
empty: { step: 4, sectionKey: 'empty_fields' },
missing: { step: 5, sectionKey: 'missing_comments' },
length: { step: 6, sectionKey: 'length_issues' },
standards: { step: 7, sectionKey: 'standard_violations' },
redundancy: { step: 2, sectionKey: 'redundancy_fields' }, // 冗余字段与合并候选同步骤(都来自 Step 2)
empty: { step: 3, sectionKey: 'empty_fields' },
missing: { step: 4, sectionKey: 'missing_comments' },
length: { step: 5, sectionKey: 'length_issues' },
standards: { step: 6, sectionKey: 'standard_violations' },
};
// 每 tab 实时状态:idle=非分析对象 / running=分析中 / done=分析完成
const tabStatus = computed(() => {
......@@ -874,8 +874,8 @@
form.steps.push(s.num);
}
}
// 兜底:报告生成不再是 Step,不在 form.steps 里出现(兼容旧本地缓存里残留的 8);Step 3 已移除
form.steps = form.steps.filter(n => n !== 8 && n !== 3);
// 兜底:报告生成不再是 Step(兼容旧缓存里残留的 8);Step 3「数据验证」已移除(兼容旧缓存里的 3);重新编号后 7 也已废弃
form.steps = form.steps.filter(n => n >= 1 && n <= 6);
try {
const r = await fetch('/api/jobs', {
method: 'POST',
......
Markdown is supported
0%
or
You are about to add 0 people to the discussion. Proceed with caution.
Finish editing this message first!
Please register or to comment