Commit 472a0c4a authored by Data Governance Dev's avatar Data Governance Dev

refactor(web): 重新编号 Step — 4-7 → 3-6,让步骤连续 1-6

移除 Step 3「数据验证」后,Step 号出现断层:1,2,4,5,6,7。重新编号让 Step 号
连续 1-6,对用户更直观。

改动:

orchestrator.py:
- STEP_REGISTRY:4→3 (空字段扫描), 5→4 (缺注释), 6→5 (字段长度), 7→6 (国标校验)
- STEP_DEPENDENCIES 同步重新编号
- step_name() 同步重新编号('3_empty_fields' / '4_missing_comments' / 等)
- _run_step4/5/6/7 重命名为 _run_step3/4/5/6(调用方 STEP_REGISTRY 已对,自动匹配)
- 顶部 docstring 重写:6 步流程 + 重新编号说明

step_impl/*.py:每个文件里的 step="X" 日志标签同步重新编号
(4→3, 5→4, 6→5, 7→6),同时把 docstring 和 message 文本里的 Step X 字样也改了。
文件本身保留旧文件名(如 step4_empty_fields.py)—— 内部实现细节,
不影响外部行为;如果未来想重命名可单独一次提交。

routes.py:list_steps() 的 StepInfo 也重新编号(4→3, 5→4, 6→5, 7→6)

frontend index.html:
- form.steps 默认 [1,2,4,5,6,7] → [1,2,3,4,5,6]
- 兜底过滤改成 n >= 1 && n <= 6(兼容旧缓存残留的 8/7/3)
- TAB_STEP_MAP:
  - redundancy tab:step 3 → 2(与 merge 同一来源)
  - empty: 4 → 3
  - missing: 5 → 4
  - length: 6 → 5
  - standards: 7 → 6

注:未触及 workflow/ 旧 CLI 目录、docs/DESIGN.md / README.md 文档,
那些是历史描述,单独一次文档 PR 一起改。

行为:UI 显示 Step 1-6 连续 6 个分析 Step,进度计算 / 报告生成
/ section 关联都不变。文件 / 函数命名差异仅是内部实现细节。
parent d2cd3ad6
...@@ -74,16 +74,16 @@ async def list_steps(): ...@@ -74,16 +74,16 @@ async def list_steps():
StepInfo(num=2, title="表合并与冗余字段分析", StepInfo(num=2, title="表合并与冗余字段分析",
description="离线:找结构高度相似的表 + 高频字段(LLM 必须:合并 verdict 与冗余分类)", description="离线:找结构高度相似的表 + 高频字段(LLM 必须:合并 verdict 与冗余分类)",
requires_db=False, llm_mode="required", required=False), requires_db=False, llm_mode="required", required=False),
StepInfo(num=4, title="大范围空字段扫描", StepInfo(num=3, title="大范围空字段扫描",
description="单表全列扫,统计 NULL/空值比例(纯程序化检查,不依赖 LLM)", description="单表全列扫,统计 NULL/空值比例(纯程序化检查,不依赖 LLM)",
requires_db=True, llm_mode="none", required=False), requires_db=True, llm_mode="none", required=False),
StepInfo(num=5, title="缺失注释字段 + 推测", StepInfo(num=4, title="缺失注释字段 + 推测",
description="离线:找无注释字段,LLM 必须:批量推测语义", description="离线:找无注释字段,LLM 必须:批量推测语义",
requires_db=False, llm_mode="required", required=False), requires_db=False, llm_mode="required", required=False),
StepInfo(num=6, title="字段长度检查", StepInfo(num=5, title="字段长度检查",
description="离线:识别字段定义长度超出国家标准(纯规则匹配,不依赖 LLM)", description="离线:识别字段定义长度超出国家标准(纯规则匹配,不依赖 LLM)",
requires_db=False, llm_mode="none", required=False), requires_db=False, llm_mode="none", required=False),
StepInfo(num=7, title="国家标准校验", StepInfo(num=6, title="国家标准校验",
description="连库抽样:身份证/USCC/手机号/行政区划符合性", description="连库抽样:身份证/USCC/手机号/行政区划符合性",
requires_db=True, llm_mode="none", required=False), requires_db=True, llm_mode="none", required=False),
]) ])
......
...@@ -2,12 +2,13 @@ ...@@ -2,12 +2,13 @@
把 6 个分析 Step 串起来执行: 把 6 个分析 Step 串起来执行:
1. 数据字典 → 2. 合并/冗余(离线) 1. 数据字典 → 2. 合并/冗余(离线)
→ 4. 空字段(连库) → 5. 缺注释(离线) → 6. 字段长度(离线) → 3. 空字段(连库) → 4. 缺注释(离线) → 5. 字段长度(离线)
→ 7. 国标校验(连库) → 6. 国标校验(连库)
注:Step 3「数据验证」已移除——它实际产出是占位(xzqh 孤儿检查、数据质量均未实现), 历史:早期版本有 Step 3「数据验证」,但产出是占位(xzqh 孤儿检查、
而且会反向覆盖 Step 2 的 LLM 分类结果(返回 section_key='redundancy_fields'), 数据质量均未实现),且会反向覆盖 Step 2 的 LLM 分类结果(返回
造成主步骤最关键的 LLM 输出被丢失。 section_key='redundancy_fields')。已移除,并顺手重新编号
4-7 → 3-6 让 Step 号连续。
报告生成不在 Step 列表内:主步骤全部跑完后无条件触发(一次性生成 报告生成不在 Step 列表内:主步骤全部跑完后无条件触发(一次性生成
Markdown + docx,供前端下载)。 Markdown + docx,供前端下载)。
...@@ -41,9 +42,9 @@ logger = logging.getLogger(__name__) ...@@ -41,9 +42,9 @@ logger = logging.getLogger(__name__)
# 顺序敏感:前面 Step 的产出可被后面 Step 复用 # 顺序敏感:前面 Step 的产出可被后面 Step 复用
# #
# llm_mode 取值: # llm_mode 取值:
# "none" —— 不用 LLM(Step 1、3、4、6、7) # "none" —— 不用 LLM(Step 1、3、5、6)
# "optional" —— LLM 可用就用,失败降级到规则推理(当前未使用) # "optional" —— LLM 可用就用,失败降级到规则推理(当前未使用)
# "required" —— 必须 LLM;缺 LLM = 任务失败(Step 2、5) # "required" —— 必须 LLM;缺 LLM = 任务失败(Step 2、4)
# #
# required 标记:true 表示用户在 UI 上不能取消勾选(如 Step 1 为全部依赖的根基) # required 标记:true 表示用户在 UI 上不能取消勾选(如 Step 1 为全部依赖的根基)
# 报告生成不再作为 Step 8 注册——它是流程结束后的固定动作,见 _generate_reports # 报告生成不再作为 Step 8 注册——它是流程结束后的固定动作,见 _generate_reports
...@@ -51,10 +52,10 @@ STEP_REGISTRY: list[tuple[int, str, Callable, bool, str, bool]] = [ ...@@ -51,10 +52,10 @@ STEP_REGISTRY: list[tuple[int, str, Callable, bool, str, bool]] = [
# (num, title, fn, requires_db, llm_mode, required) # (num, title, fn, requires_db, llm_mode, required)
(1, "获取数据字典", "_run_step1", True, "none", True), (1, "获取数据字典", "_run_step1", True, "none", True),
(2, "表合并与冗余字段分析", "_run_step2", False, "required", False), (2, "表合并与冗余字段分析", "_run_step2", False, "required", False),
(4, "大范围空字段扫描", "_run_step4", True, "none", False), (3, "大范围空字段扫描", "_run_step3", True, "none", False),
(5, "缺失注释检查 + 推测", "_run_step5", False, "required", False), (4, "缺失注释检查 + 推测", "_run_step4", False, "required", False),
(6, "字段长度检查", "_run_step6", False, "none", False), (5, "字段长度检查", "_run_step5", False, "none", False),
(7, "国家标准校验", "_run_step7", True, "none", False), (6, "国家标准校验", "_run_step6", True, "none", False),
] ]
# ── Step 依赖图 ─────────────────────────────────────────── # ── Step 依赖图 ───────────────────────────────────────────
...@@ -64,10 +65,10 @@ STEP_REGISTRY: list[tuple[int, str, Callable, bool, str, bool]] = [ ...@@ -64,10 +65,10 @@ STEP_REGISTRY: list[tuple[int, str, Callable, bool, str, bool]] = [
STEP_DEPENDENCIES: dict[int, set[int]] = { STEP_DEPENDENCIES: dict[int, set[int]] = {
1: set(), # 根基 1: set(), # 根基
2: {1}, 2: {1},
4: set(), # 空字段扫描内部自取数据,可与 Step 1 并发 3: set(), # 空字段扫描内部自取数据,可与 Step 1 并发
4: {1},
5: {1}, 5: {1},
6: {1}, 6: {1},
7: {1},
} }
# ── 关键 Step 集合 ──────────────────────────────────────── # ── 关键 Step 集合 ────────────────────────────────────────
...@@ -392,10 +393,10 @@ def step_name(num: int) -> str: ...@@ -392,10 +393,10 @@ def step_name(num: int) -> str:
return { return {
1: "1_data_dict", 1: "1_data_dict",
2: "2_merge_redundancy", 2: "2_merge_redundancy",
4: "4_empty_fields", 3: "3_empty_fields",
5: "5_missing_comments", 4: "4_missing_comments",
6: "6_length_check", 5: "5_length_check",
7: "7_standards", 6: "6_standards",
}[num] }[num]
...@@ -415,31 +416,31 @@ def _run_step2(cfg, step_outputs, llm, findings_dir, log, cancel_event): ...@@ -415,31 +416,31 @@ def _run_step2(cfg, step_outputs, llm, findings_dir, log, cancel_event):
return {"section_key": "merge_candidates", "data": data} return {"section_key": "merge_candidates", "data": data}
def _run_step4(cfg, step_outputs, llm, findings_dir, log, cancel_event): def _run_step3(cfg, step_outputs, llm, findings_dir, log, cancel_event):
"""Step 4: 空字段扫描(连库)—— 复用 Step 1 已拿到的数据字典""" """Step 3: 空字段扫描(连库)—— 复用 Step 1 已拿到的数据字典"""
from .step_impl.step4_empty_fields import run_step4 from .step_impl.step4_empty_fields import run_step4
dict_data = step_outputs.get("1_data_dict") or {} dict_data = step_outputs.get("1_data_dict") or {}
data = run_step4(cfg, dict_data=dict_data, log=log) data = run_step4(cfg, dict_data=dict_data, log=log)
return {"section_key": "empty_fields", "data": data} return {"section_key": "empty_fields", "data": data}
def _run_step5(cfg, step_outputs, llm, findings_dir, log, cancel_event): def _run_step4(cfg, step_outputs, llm, findings_dir, log, cancel_event):
"""Step 5: 缺注释 + LLM 推测""" """Step 4: 缺注释 + LLM 推测"""
from .step_impl.step5_missing_comments import run_step5 from .step_impl.step5_missing_comments import run_step5
data = run_step5(dict_data=step_outputs.get("1_data_dict", {}), data = run_step5(dict_data=step_outputs.get("1_data_dict", {}),
llm=llm, log=log) llm=llm, log=log)
return {"section_key": "missing_comments", "data": data} return {"section_key": "missing_comments", "data": data}
def _run_step6(cfg, step_outputs, llm, findings_dir, log, cancel_event): def _run_step5(cfg, step_outputs, llm, findings_dir, log, cancel_event):
"""Step 6: 字段长度检查(离线)""" """Step 5: 字段长度检查(离线)"""
from .step_impl.step6_length_check import run_step6 from .step_impl.step6_length_check import run_step6
data = run_step6(dict_data=step_outputs.get("1_data_dict", {}), log=log) data = run_step6(dict_data=step_outputs.get("1_data_dict", {}), log=log)
return {"section_key": "length_issues", "data": data} return {"section_key": "length_issues", "data": data}
def _run_step7(cfg, step_outputs, llm, findings_dir, log, cancel_event): def _run_step6(cfg, step_outputs, llm, findings_dir, log, cancel_event):
"""Step 7: 国标校验(连库)""" """Step 6: 国标校验(连库)"""
from .step_impl.step7_standards import run_step7 from .step_impl.step7_standards import run_step7
data = run_step7(cfg, dict_data=step_outputs.get("1_data_dict", {}), log=log) data = run_step7(cfg, dict_data=step_outputs.get("1_data_dict", {}), log=log)
return {"section_key": "standard_violations", "data": data} return {"section_key": "standard_violations", "data": data}
......
"""Step 4: 大范围空字段扫描(纯程序化检查) """Step 3: 大范围空字段扫描(纯程序化检查)
对每张表跑动态 SQL 统计: 对每张表跑动态 SQL 统计:
- COUNT(*) 总行数 - COUNT(*) 总行数
...@@ -39,23 +39,23 @@ def run_step4(cfg: DBConfig, log: Callable | None = None, ...@@ -39,23 +39,23 @@ def run_step4(cfg: DBConfig, log: Callable | None = None,
""" """
if log: if log:
log("INFO", log("INFO",
f"Step 4 开始扫描空字段 (high≥{high_threshold:.0%}, mid≥{mid_threshold:.0%}, min_rows={min_rows_to_check})", f"Step 3 开始扫描空字段 (high≥{high_threshold:.0%}, mid≥{mid_threshold:.0%}, min_rows={min_rows_to_check})",
step="4") step="3")
if dict_data and dict_data.get("data_dictionary"): if dict_data and dict_data.get("data_dictionary"):
# 复用 Step 1 产出(Wave 调度器已保证 Step 1 先完成) # 复用 Step 1 产出(Wave 调度器已保证 Step 1 先完成)
columns = dict_data["data_dictionary"] columns = dict_data["data_dictionary"]
table_summary = dict_data.get("table_summary", []) table_summary = dict_data.get("table_summary", [])
if log: if log:
log("INFO", " · 复用 Step 1 数据字典", step="4") log("INFO", " · 复用 Step 1 数据字典", step="3")
else: else:
# 兜底:没拿到 Step 1 产出时回退自取(独立跑 Step 4 时用得到) # 兜底:没拿到 Step 1 产出时回退自取(独立跑 Step 3 时用得到)
from .step1_data_dict import run_step1 from .step1_data_dict import run_step1
dict_data = run_step1(cfg, log=lambda lvl, msg, **_kw: log(lvl, msg, step="4") if log else None) dict_data = run_step1(cfg, log=lambda lvl, msg, **_kw: log(lvl, msg, step="3") if log else None)
columns = dict_data["data_dictionary"] columns = dict_data["data_dictionary"]
table_summary = dict_data.get("table_summary", []) table_summary = dict_data.get("table_summary", [])
if log: if log:
log("INFO", " · 独立拉取数据字典(未复用 Step 1)", step="4") log("INFO", " · 独立拉取数据字典(未复用 Step 1)", step="3")
by_table: dict[str, list[dict]] = defaultdict(list) by_table: dict[str, list[dict]] = defaultdict(list)
for row in columns: for row in columns:
...@@ -88,15 +88,15 @@ def run_step4(cfg: DBConfig, log: Callable | None = None, ...@@ -88,15 +88,15 @@ def run_step4(cfg: DBConfig, log: Callable | None = None,
table_names = sorted(by_table.keys()) table_names = sorted(by_table.keys())
total = len(table_names) total = len(table_names)
if log: if log:
log("INFO", f"将扫描 {total} 张表(已过滤 TEXT_TYPES)", step="4") log("INFO", f"将扫描 {total} 张表(已过滤 TEXT_TYPES)", step="3")
try: try:
with open_db(cfg) as db: with open_db(cfg) as db:
for idx, table in enumerate(table_names, 1): for idx, table in enumerate(table_names, 1):
if log and idx % 10 == 0: if log and idx % 10 == 0:
log("INFO", f" 进度 {idx}/{total} - 当前表: {table}", step="4") log("INFO", f" 进度 {idx}/{total} - 当前表: {table}", step="3")
elif log and idx == 1: elif log and idx == 1:
log("INFO", f" 进度 1/{total} - 当前表: {table}", step="4") log("INFO", f" 进度 1/{total} - 当前表: {table}", step="3")
# 拿实际行数(避免 TABLE_ROWS 估算不准) # 拿实际行数(避免 TABLE_ROWS 估算不准)
try: try:
...@@ -109,7 +109,7 @@ def run_step4(cfg: DBConfig, log: Callable | None = None, ...@@ -109,7 +109,7 @@ def run_step4(cfg: DBConfig, log: Callable | None = None,
except Exception as e: except Exception as e:
skipped.append({"table": table, "reason": f"统计行数失败: {e}"}) skipped.append({"table": table, "reason": f"统计行数失败: {e}"})
if log: if log:
log("WARN", f" · {table} 行数统计失败: {e}", step="4") log("WARN", f" · {table} 行数统计失败: {e}", step="3")
continue continue
if row_count < min_rows_to_check: if row_count < min_rows_to_check:
...@@ -134,7 +134,7 @@ def run_step4(cfg: DBConfig, log: Callable | None = None, ...@@ -134,7 +134,7 @@ def run_step4(cfg: DBConfig, log: Callable | None = None,
except Exception as e: except Exception as e:
skipped.append({"table": table, "reason": f"扫描失败: {e}"}) skipped.append({"table": table, "reason": f"扫描失败: {e}"})
if log: if log:
log("WARN", f" · {table} 空字段扫描失败: {e}", step="4") log("WARN", f" · {table} 空字段扫描失败: {e}", step="3")
continue continue
if not row: if not row:
...@@ -189,8 +189,8 @@ def run_step4(cfg: DBConfig, log: Callable | None = None, ...@@ -189,8 +189,8 @@ def run_step4(cfg: DBConfig, log: Callable | None = None,
logger.debug(f"表 {table}: 高空 {len(high_in_table)}, 中空 {len(mid_in_table)}") logger.debug(f"表 {table}: 高空 {len(high_in_table)}, 中空 {len(mid_in_table)}")
except Exception as e: except Exception as e:
if log: if log:
log("ERROR", f"Step 4 数据库扫描失败: {type(e).__name__}: {e}", step="4") log("ERROR", f"Step 3 数据库扫描失败: {type(e).__name__}: {e}", step="3")
logger.exception("Step 4 详细异常") logger.exception("Step 3 详细异常")
top_tables.sort(key=lambda x: x["high_empty_field_count"], reverse=True) top_tables.sort(key=lambda x: x["high_empty_field_count"], reverse=True)
...@@ -210,7 +210,7 @@ def run_step4(cfg: DBConfig, log: Callable | None = None, ...@@ -210,7 +210,7 @@ def run_step4(cfg: DBConfig, log: Callable | None = None,
log("INFO", log("INFO",
f"扫描完成: 高空 {high_count}, 中空 {mid_count}, 涉及 {len(tables_with_issues)} 张表, " f"扫描完成: 高空 {high_count}, 中空 {mid_count}, 涉及 {len(tables_with_issues)} 张表, "
f"跳过 {len(skipped)} 张", f"跳过 {len(skipped)} 张",
step="4") step="3")
return { return {
"summary": summary, "summary": summary,
......
"""Step 5: 缺失注释字段检查 + LLM 推测(必跑 LLM) """Step 4: 缺失注释字段检查 + LLM 推测(必跑 LLM)
字段注释的语义判断本质上需要 LLM: 字段注释的语义判断本质上需要 LLM:
- 拼音首字母缩写(如 xzqhbm)、业务缩写、英文组合 → 必须由 LLM 翻译 - 拼音首字母缩写(如 xzqhbm)、业务缩写、英文组合 → 必须由 LLM 翻译
...@@ -27,11 +27,11 @@ def run_step5(dict_data: dict, llm: LLMClient | None = None, ...@@ -27,11 +27,11 @@ def run_step5(dict_data: dict, llm: LLMClient | None = None,
columns = dict_data.get("data_dictionary", []) columns = dict_data.get("data_dictionary", [])
if not columns: if not columns:
if log: if log:
log("WARN", "未获取到任何字段元数据,跳过 Step 5", step="5") log("WARN", "未获取到任何字段元数据,跳过 Step 4", step="4")
return {"summary": {}, "by_table": [], "predicted_comments": [], "unpredictable_sample": []} return {"summary": {}, "by_table": [], "predicted_comments": [], "unpredictable_sample": []}
if log: if log:
log("INFO", f"待检查字段总数: {len(columns)}", step="5") log("INFO", f"待检查字段总数: {len(columns)}", step="4")
# 1. 收集所有无注释字段,按 (table_name, table_comment) 分组 # 1. 收集所有无注释字段,按 (table_name, table_comment) 分组
missing = [] missing = []
...@@ -60,12 +60,12 @@ def run_step5(dict_data: dict, llm: LLMClient | None = None, ...@@ -60,12 +60,12 @@ def run_step5(dict_data: dict, llm: LLMClient | None = None,
if log: if log:
log("INFO", log("INFO",
f" · 缺失注释字段: {len(missing)} 个 (覆盖 {len(by_table)} 张表)", f" · 缺失注释字段: {len(missing)} 个 (覆盖 {len(by_table)} 张表)",
step="5") step="4")
# 2. 按表分批调用 LLM(必跑;单批失败会让任务终止) # 2. 按表分批调用 LLM(必跑;单批失败会让任务终止)
if not missing: if not missing:
if log: if log:
log("INFO", " · 无缺失注释字段,跳过 LLM", step="5") log("INFO", " · 无缺失注释字段,跳过 LLM", step="4")
return _empty_result(by_table) return _empty_result(by_table)
BATCH = 30 BATCH = 30
...@@ -81,7 +81,7 @@ def run_step5(dict_data: dict, llm: LLMClient | None = None, ...@@ -81,7 +81,7 @@ def run_step5(dict_data: dict, llm: LLMClient | None = None,
log("INFO", log("INFO",
f" · LLM 推测 [{llm_called}] {tname} " f" · LLM 推测 [{llm_called}] {tname} "
f"({chunk_idx}/{len(chunks)} 批, {len(chunk)} 个字段)", f"({chunk_idx}/{len(chunks)} 批, {len(chunk)} 个字段)",
step="5") step="4")
try: try:
results = llm.predict_field_comments_batch( results = llm.predict_field_comments_batch(
table_name=tname, table_name=tname,
...@@ -94,8 +94,8 @@ def run_step5(dict_data: dict, llm: LLMClient | None = None, ...@@ -94,8 +94,8 @@ def run_step5(dict_data: dict, llm: LLMClient | None = None,
if log: if log:
log("ERROR", log("ERROR",
f" · LLM 推测失败({tname} 第 {chunk_idx} 批): {e}(任务将终止)", f" · LLM 推测失败({tname} 第 {chunk_idx} 批): {e}(任务将终止)",
step="5") step="4")
logger.exception("Step 5 LLM 推测批次失败") logger.exception("Step 4 LLM 推测批次失败")
raise raise
# 把 LLM 结果写回 entry # 把 LLM 结果写回 entry
...@@ -121,11 +121,11 @@ def run_step5(dict_data: dict, llm: LLMClient | None = None, ...@@ -121,11 +121,11 @@ def run_step5(dict_data: dict, llm: LLMClient | None = None,
log("INFO", log("INFO",
f" · LLM 命中 {len(predicted)}/{len(missing)}, " f" · LLM 命中 {len(predicted)}/{len(missing)}, "
f"LLM 解析失败 {len(unpredicted)}", f"LLM 解析失败 {len(unpredicted)}",
step="5") step="4")
log("INFO", log("INFO",
f"缺失注释 {len(missing)} 个字段,覆盖 {len(by_table)} 张表;" f"缺失注释 {len(missing)} 个字段,覆盖 {len(by_table)} 张表;"
f"LLM 推测成功 {len(predicted)}", f"LLM 推测成功 {len(predicted)}",
step="5") step="4")
return { return {
"summary": { "summary": {
......
"""Step 6: 字段长度检查(纯规则匹配) """Step 5: 字段长度检查(纯规则匹配)
识别字段定义长度超出标准所需(如身份证号 18 位用 varchar(50) 存储)。 识别字段定义长度超出标准所需(如身份证号 18 位用 varchar(50) 存储)。
仅基于内置规则匹配,不依赖 LLM。 仅基于内置规则匹配,不依赖 LLM。
...@@ -156,13 +156,13 @@ def run_step6(dict_data: dict, log: Callable | None = None) -> dict: ...@@ -156,13 +156,13 @@ def run_step6(dict_data: dict, log: Callable | None = None) -> dict:
columns = dict_data.get("data_dictionary", []) columns = dict_data.get("data_dictionary", [])
if not columns: if not columns:
if log: if log:
log("WARN", "未获取到任何字段元数据,跳过 Step 6", step="6") log("WARN", "未获取到任何字段元数据,跳过 Step 5", step="5")
return {"summary": {"total_issues": 0}, "issues": []} return {"summary": {"total_issues": 0}, "issues": []}
if log: if log:
log("INFO", log("INFO",
f"内置长度规则: {len(LENGTH_RULES)} 条, 待扫描字段: {len(columns)}", f"内置长度规则: {len(LENGTH_RULES)} 条, 待扫描字段: {len(columns)}",
step="6") step="5")
issues = [] issues = []
match_by_name = 0 match_by_name = 0
...@@ -209,7 +209,7 @@ def run_step6(dict_data: dict, log: Callable | None = None) -> dict: ...@@ -209,7 +209,7 @@ def run_step6(dict_data: dict, log: Callable | None = None) -> dict:
log("INFO", log("INFO",
f"规则匹配: 字段名 {match_by_name} 次 + 注释关键字 {match_by_comment} 次," f"规则匹配: 字段名 {match_by_name} 次 + 注释关键字 {match_by_comment} 次,"
f"发现异常 {len(issues)} 条", f"发现异常 {len(issues)} 条",
step="6") step="5")
return { return {
"summary": { "summary": {
......
"""Step 7: 国家标准字段校验(连库) """Step 6: 国家标准字段校验(连库)
按字段名匹配 standards 插件,抽样校验实际数据: 按字段名匹配 standards 插件,抽样校验实际数据:
- 身份证号(GB 11643) - 身份证号(GB 11643)
...@@ -46,7 +46,7 @@ SAMPLE_SPECS: list[dict] = [ ...@@ -46,7 +46,7 @@ SAMPLE_SPECS: list[dict] = [
def run_step7(cfg: DBConfig, dict_data: dict, log: Callable | None = None) -> dict: def run_step7(cfg: DBConfig, dict_data: dict, log: Callable | None = None) -> dict:
"""执行 Step 7:国标校验""" """执行 Step 6:国标校验"""
columns = dict_data.get("data_dictionary", []) columns = dict_data.get("data_dictionary", [])
if not columns: if not columns:
return {"summary": {}, "violations_by_standard": [], "violations": []} return {"summary": {}, "violations_by_standard": [], "violations": []}
...@@ -67,7 +67,7 @@ def run_step7(cfg: DBConfig, dict_data: dict, log: Callable | None = None) -> di ...@@ -67,7 +67,7 @@ def run_step7(cfg: DBConfig, dict_data: dict, log: Callable | None = None) -> di
if not standards_by_field: if not standards_by_field:
if log: if log:
log("WARN", "未发现任何标准插件", step="7") log("WARN", "未发现任何标准插件", step="6")
return { return {
"summary": {"standards_applied": 0, "fields_checked": 0, "violations": 0}, "summary": {"standards_applied": 0, "fields_checked": 0, "violations": 0},
"violations_by_standard": [], "violations_by_standard": [],
...@@ -79,7 +79,7 @@ def run_step7(cfg: DBConfig, dict_data: dict, log: Callable | None = None) -> di ...@@ -79,7 +79,7 @@ def run_step7(cfg: DBConfig, dict_data: dict, log: Callable | None = None) -> di
f"已加载 {len(all_standards_info)} 个标准插件, " f"已加载 {len(all_standards_info)} 个标准插件, "
f"匹配字段 {len(standards_by_field)} 个, " f"匹配字段 {len(standards_by_field)} 个, "
f"最大每字段抽样 {SAMPLE_LIMIT} 行 / 每字段最多 {MAX_TABLES_PER_FIELD} 张表", f"最大每字段抽样 {SAMPLE_LIMIT} 行 / 每字段最多 {MAX_TABLES_PER_FIELD} 张表",
step="7") step="6")
loader = get_sql_loader() loader = get_sql_loader()
...@@ -90,7 +90,7 @@ def run_step7(cfg: DBConfig, dict_data: dict, log: Callable | None = None) -> di ...@@ -90,7 +90,7 @@ def run_step7(cfg: DBConfig, dict_data: dict, log: Callable | None = None) -> di
if log: if log:
log("DEBUG", log("DEBUG",
f" · [{std_idx}/{len(standards_by_field)}] 标准字段 {std_field} 未出现在任何表中, 跳过", f" · [{std_idx}/{len(standards_by_field)}] 标准字段 {std_field} 未出现在任何表中, 跳过",
step="7") step="6")
continue continue
tables_with_field = by_name[std_field][:MAX_TABLES_PER_FIELD] tables_with_field = by_name[std_field][:MAX_TABLES_PER_FIELD]
std_id = std_instance.standard_id std_id = std_instance.standard_id
...@@ -107,7 +107,7 @@ def run_step7(cfg: DBConfig, dict_data: dict, log: Callable | None = None) -> di ...@@ -107,7 +107,7 @@ def run_step7(cfg: DBConfig, dict_data: dict, log: Callable | None = None) -> di
log("INFO", log("INFO",
f" · [{std_idx}/{len(standards_by_field)}] " f" · [{std_idx}/{len(standards_by_field)}] "
f"{std_id} ({std_field}) → {len(tables_with_field)} 张表", f"{std_id} ({std_field}) → {len(tables_with_field)} 张表",
step="7") step="6")
for col_record in tables_with_field: for col_record in tables_with_field:
table = col_record["table_name"] table = col_record["table_name"]
...@@ -139,7 +139,7 @@ def run_step7(cfg: DBConfig, dict_data: dict, log: Callable | None = None) -> di ...@@ -139,7 +139,7 @@ def run_step7(cfg: DBConfig, dict_data: dict, log: Callable | None = None) -> di
if log: if log:
log("WARN", log("WARN",
f" - {table}.{field_name} 抽样失败: {e}", f" - {table}.{field_name} 抽样失败: {e}",
step="7") step="6")
continue continue
sampled = len(rows) sampled = len(rows)
...@@ -188,8 +188,8 @@ def run_step7(cfg: DBConfig, dict_data: dict, log: Callable | None = None) -> di ...@@ -188,8 +188,8 @@ def run_step7(cfg: DBConfig, dict_data: dict, log: Callable | None = None) -> di
) )
except Exception as e: except Exception as e:
if log: if log:
log("ERROR", f"Step 7 连库校验失败: {type(e).__name__}: {e}", step="7") log("ERROR", f"Step 6 连库校验失败: {type(e).__name__}: {e}", step="6")
logger.exception("Step 7 详细异常") logger.exception("Step 6 详细异常")
by_standard_list = [] by_standard_list = []
for std_id, rec in by_standard.items(): for std_id, rec in by_standard.items():
...@@ -198,7 +198,7 @@ def run_step7(cfg: DBConfig, dict_data: dict, log: Callable | None = None) -> di ...@@ -198,7 +198,7 @@ def run_step7(cfg: DBConfig, dict_data: dict, log: Callable | None = None) -> di
if log: if log:
log("INFO", log("INFO",
f" · 标准 {rec['standard']} 抽样 {rec['total_sampled']} 条全部合规,跳过", f" · 标准 {rec['standard']} 抽样 {rec['total_sampled']} 条全部合规,跳过",
step="7") step="6")
continue continue
violating_fields = [d for d in rec["fields"] if d.get("violations", 0) > 0] violating_fields = [d for d in rec["fields"] if d.get("violations", 0) > 0]
by_standard_list.append({ by_standard_list.append({
...@@ -219,7 +219,7 @@ def run_step7(cfg: DBConfig, dict_data: dict, log: Callable | None = None) -> di ...@@ -219,7 +219,7 @@ def run_step7(cfg: DBConfig, dict_data: dict, log: Callable | None = None) -> di
}) })
if log: if log:
log("INFO", f"校验完成: {len(by_standard_list)} 个标准有违规, {total_fields_checked} 个字段, {total_violations} 条违规", step="7") log("INFO", f"校验完成: {len(by_standard_list)} 个标准有违规, {total_fields_checked} 个字段, {total_violations} 条违规", step="6")
return { return {
"summary": { "summary": {
......
...@@ -732,7 +732,7 @@ ...@@ -732,7 +732,7 @@
database: 'smart-build', database: 'smart-build',
charset: 'utf8mb4', charset: 'utf8mb4',
connect_timeout: 10, connect_timeout: 10,
steps: [1, 2, 4, 5, 6, 7], steps: [1, 2, 3, 4, 5, 6],
}); });
const testing = ref(false); const testing = ref(false);
...@@ -773,11 +773,11 @@ ...@@ -773,11 +773,11 @@
// tab → (所属 step, 对应 sections 里的 key) // tab → (所属 step, 对应 sections 里的 key)
const TAB_STEP_MAP = { const TAB_STEP_MAP = {
merge: { step: 2, sectionKey: 'merge_candidates' }, merge: { step: 2, sectionKey: 'merge_candidates' },
redundancy: { step: 3, sectionKey: 'redundancy_fields' }, redundancy: { step: 2, sectionKey: 'redundancy_fields' }, // 冗余字段与合并候选同步骤(都来自 Step 2)
empty: { step: 4, sectionKey: 'empty_fields' }, empty: { step: 3, sectionKey: 'empty_fields' },
missing: { step: 5, sectionKey: 'missing_comments' }, missing: { step: 4, sectionKey: 'missing_comments' },
length: { step: 6, sectionKey: 'length_issues' }, length: { step: 5, sectionKey: 'length_issues' },
standards: { step: 7, sectionKey: 'standard_violations' }, standards: { step: 6, sectionKey: 'standard_violations' },
}; };
// 每 tab 实时状态:idle=非分析对象 / running=分析中 / done=分析完成 // 每 tab 实时状态:idle=非分析对象 / running=分析中 / done=分析完成
const tabStatus = computed(() => { const tabStatus = computed(() => {
...@@ -874,8 +874,8 @@ ...@@ -874,8 +874,8 @@
form.steps.push(s.num); form.steps.push(s.num);
} }
} }
// 兜底:报告生成不再是 Step,不在 form.steps 里出现(兼容旧本地缓存里残留的 8);Step 3 已移除 // 兜底:报告生成不再是 Step(兼容旧缓存里残留的 8);Step 3「数据验证」已移除(兼容旧缓存里的 3);重新编号后 7 也已废弃
form.steps = form.steps.filter(n => n !== 8 && n !== 3); form.steps = form.steps.filter(n => n >= 1 && n <= 6);
try { try {
const r = await fetch('/api/jobs', { const r = await fetch('/api/jobs', {
method: 'POST', method: 'POST',
......
Markdown is supported
0%
or
You are about to add 0 people to the discussion. Proceed with caution.
Finish editing this message first!
Please register or to comment