Commit ba75ed1b authored by Data Governance Dev's avatar Data Governance Dev

feat(standards): IND-001 a/b/c/d 合并为单分析项目 + 4 子检查合一

需求:勾一个 IND-001 checkbox 跑全部 4 子检查;
结果合并展示,单条违规里显示「哪些子检查失败」。

改动:
- step7_standards.py: 新增 _IND_001_SUB_IDS / _merge_violations_by_tuple
  / _wrap_combined_ind_001 / run_step7_for_ind_001,单 (table,col,value)
  多子检查失败行合并,rule_type 列改为 'IND-001-a, IND-001-b',
  error 列改为 '[a] 长度 19 ≠ 18\n[b] 校验位错误' 拼接原文
- orchestrator.py: 注册 std_ind_001 step(order=100,llm=none),
  跳过 IND-001-a/b/c/d 自动注册,避免 4 个独立 section 还在
- analysis_tree.json: 国标字段规范组删 4 行 IND-001-a/b/c/d,加 1 行 std_ind_001
- routes.py: /api/match-config 过滤未注册为 step 的子 indicator;
  把 IND-001 a/b/c/d 的 applies_to_fields / comment_keywords 聚合到 std_ind_001
  的输入框,避免合并后前端输入框丢失
- 修复 double-wrap bug:_run_standards_ind_001 不再二次包裹
  run_step7_for_ind_001 原返回值(已含 {section_key, data})
- WORKLOG 追加 4 段:合并设计 / 关键词输入框回归 / 标题 '合并 a/b/c/d' 去除 /
  结果不显示修复(double-wrap 根因 + 烟测)
parent b968bb24
...@@ -48,3 +48,4 @@ LLM 不是「增强」,对部分步骤是核心依赖: ...@@ -48,3 +48,4 @@ LLM 不是「增强」,对部分步骤是核心依赖:
- 每做一个任务就在工作记录中记录一次 - 每做一个任务就在工作记录中记录一次
- 不要自动提交,我说提交再提交 - 不要自动提交,我说提交再提交
- 每次提交要写清除修改内容 - 每次提交要写清除修改内容
- 每次踩坑记录一下
This diff is collapsed.
...@@ -216,9 +216,16 @@ async def get_match_config(): ...@@ -216,9 +216,16 @@ async def get_match_config():
# ── Step 7:每个 indicator 的 step_id(如 "std_ind_001_a")──── # ── Step 7:每个 indicator 的 step_id(如 "std_ind_001_a")────
# YAML 配置里的 key 是 standard_id(如 "IND-001-a"),需要 map 到 step_id。 # YAML 配置里的 key 是 standard_id(如 "IND-001-a"),需要 map 到 step_id。
# 约定:step_id = "std_ind_" + standard_id.lower().replace("-", "_") # 约定:step_id = "std_ind_" + standard_id.lower().replace("-", "_")
# 只输出「实际注册为 orchestrator step」的 indicator(IND-001-a/b/c/d
# 已合并到 std_ind_001,不再单独注册 —— 见下方「合并 step」块)
from ..core.orchestrator import get_step_defs as _get_step_defs
registered_step_ids = {s.step_id for s in _get_step_defs()}
for sid, meta in getattr(_list_std_step_ids_safe(), "__iter__", lambda: [])() or []: for sid, meta in getattr(_list_std_step_ids_safe(), "__iter__", lambda: [])() or []:
# meta: {"id": "IND-001-a", "name": "...", ...} # meta: {"id": "IND-001-a", "name": "...", ...}
std_id = meta.get("id", "") std_id = meta.get("id", "")
# 跳过未注册为 step 的子 indicator(合并 step 会单独处理)
if sid not in registered_step_ids:
continue
rec = step7.get(std_id) or {} rec = step7.get(std_id) or {}
if rec.get("_skip"): if rec.get("_skip"):
by_step[sid] = {"names": "", "comments": "", "skip": True} by_step[sid] = {"names": "", "comments": "", "skip": True}
...@@ -257,6 +264,30 @@ async def get_match_config(): ...@@ -257,6 +264,30 @@ async def get_match_config():
"skip": False, "skip": False,
} }
# ── 合并 step:std_ind_001(IND-001 a/b/c/d 聚合)───────────
# 4 个子 indicator 的 YAML 配置聚合到 1 个 step 输入框。
# 用户编辑的 override 在后端 run_step7_for_ind_001 里会下推到各子检查。
ind001_sub_ids = ("IND-001-a", "IND-001-b", "IND-001-c", "IND-001-d")
ind001_names: list[str] = []
ind001_comments: list[str] = []
ind001_seen_n: set = set()
ind001_seen_c: set = set()
for sub_id in ind001_sub_ids:
rec = step7.get(sub_id) or {}
for n in (rec.get("applies_to_fields") or []):
if n not in ind001_seen_n:
ind001_names.append(n)
ind001_seen_n.add(n)
for c in (rec.get("comment_keywords") or []):
if c not in ind001_seen_c:
ind001_comments.append(c)
ind001_seen_c.add(c)
by_step["std_ind_001"] = {
"names": ", ".join(ind001_names),
"comments": ", ".join(ind001_comments),
"skip": False,
}
# ── 其余 step(merge / empty / missing_comments 等)无需输入框 ── # ── 其余 step(merge / empty / missing_comments 等)无需输入框 ──
# 前端按 step_id 找不到时按 skip=True 处理即可 # 前端按 step_id 找不到时按 skip=True 处理即可
......
...@@ -18,10 +18,7 @@ ...@@ -18,10 +18,7 @@
"description": "IND-001 ~ IND-005 系列:按 GB / GA 标准做字段值级别合规校验(格式 / 校验位 / 出生日期 / 15 位兼容 / 号段 / 编码存在性 / 固定电话 / 老代码兼容)", "description": "IND-001 ~ IND-005 系列:按 GB / GA 标准做字段值级别合规校验(格式 / 校验位 / 出生日期 / 15 位兼容 / 号段 / 编码存在性 / 固定电话 / 老代码兼容)",
"default_expand": true, "default_expand": true,
"children": [ "children": [
{ "step_id": "std_ind_001_a" }, { "step_id": "std_ind_001" },
{ "step_id": "std_ind_001_b" },
{ "step_id": "std_ind_001_c" },
{ "step_id": "std_ind_001_d" },
{ "step_id": "std_ind_002_a" }, { "step_id": "std_ind_002_a" },
{ "step_id": "std_ind_002_b" }, { "step_id": "std_ind_002_b" },
{ "step_id": "std_ind_002_c" }, { "step_id": "std_ind_002_c" },
......
...@@ -540,8 +540,12 @@ def _register_indicator_steps() -> None: ...@@ -540,8 +540,12 @@ def _register_indicator_steps() -> None:
过滤: 过滤:
- 跳过 std_id 以 "STD-" 开头的已拆分旧插件(applies_to_fields=[]) - 跳过 std_id 以 "STD-" 开头的已拆分旧插件(applies_to_fields=[])
- 跳过 IND-001-a / -b / -c / -d(已被合并到 `std_ind_001` 复合 step,2026-08-12 起)
- 保留 IND-301 / IND-302 等无 applies_to_fields 的跨字段 indicator - 保留 IND-301 / IND-302 等无 applies_to_fields 的跨字段 indicator
""" """
# IND-001 的 4 个子指标合并为一个;其余 indicator 仍按 1 个 step / 1 个 tab 注册
_IND_001_SUBS = ("IND-001-a", "IND-001-b", "IND-001-c", "IND-001-d")
prev_group: str | None = None prev_group: str | None = None
same_group_count = 0 same_group_count = 0
base_order: int = 0 base_order: int = 0
...@@ -551,6 +555,12 @@ def _register_indicator_steps() -> None: ...@@ -551,6 +555,12 @@ def _register_indicator_steps() -> None:
if std_id.startswith("STD-"): if std_id.startswith("STD-"):
logger.info(f"跳过已 deprecated 旧插件: {std_id}(已被 IND-* 拆分)") logger.info(f"跳过已 deprecated 旧插件: {std_id}(已被 IND-* 拆分)")
continue continue
# 跳过 IND-001 子指标(已合并到 std_ind_001 复合 step)
if std_id in _IND_001_SUBS:
logger.info(
f"跳过子 indicator: {std_id}(已合并到 std_ind_001 复合 step)"
)
continue
group = meta.get("group", "通用") group = meta.get("group", "通用")
if group != prev_group: if group != prev_group:
same_group_count = 0 same_group_count = 0
...@@ -590,6 +600,48 @@ def _register_indicator_steps() -> None: ...@@ -590,6 +600,48 @@ def _register_indicator_steps() -> None:
_register_indicator_steps() _register_indicator_steps()
# ── IND-001 复合 step(合并 a/b/c/d 4 个子检查) ──────────────
# 2026-08-12 起:UI 上只展示 1 个 checkbox「IND-001 · 身份证号」,
# 后端内部跑全部 4 个子检查,结果按 (table, column, value) 合并展示,
# 同一个值若同时违反多个子检查 → 合并成 1 行,error 字段拼所有子错误。
# scope 限定 IND-001,其余 a/b 后缀 indicator 暂不动。
def _run_standards_ind_001(*, cfg, dict_data, llm, log, cancel_event, table_filter, match_overrides): # noqa: ARG001
from .step_impl.step7_standards import run_step7_for_ind_001
# run_step7_for_ind_001 已返回 {section_key, data},直接透传(不要再 wrap)
return run_step7_for_ind_001(
cfg, dict_data=dict_data, log=log,
table_filter=table_filter, match_overrides=match_overrides,
)
register_step(
step_id="std_ind_001",
title="IND-001 · 身份证号校验",
description=(
"GB 11643-1999 身份证号全维度校验:4 子检查合一(a 格式 / b 校验位 / c 出生日期 / d 15 位老证)。"
"同一字段同一值多子检查失败的会合并成 1 行展示,error 字段拼全部子错误信息。"
),
requires_db=True, llm_mode="none", required=False, order=100, # 国标字段规范 group 起始 100;合并 step 置首位
fn=_run_standards_ind_001,
detail=StepDetail(
purpose=(
"对身份证号字段跑 IND-001-a(格式)、IND-001-b(校验位)、"
"IND-001-c(出生日期)、IND-001-d(15 位老证)4 个子检查,"
"合并输出到 1 个 tab,相同 (table, column, value) 的多子失败合并为 1 行,"
"error 字段拼接所有子错误信息。"
),
target="字段名含 id_card / id_card_no / id_number / identity_card 或注释匹配的字段",
check=(
"a) 18 位 + 6+8+3+1 字符结构\n"
"b) ISO 7064 MOD 11-2 校验位\n"
"c) value[6:14] 解析为合法日期(含闰年 2-29)\n"
"d) 15 位老证提示(已自 1999-07-01 停发)"
),
format="GB 11643-1999(合并)/ GB 11643-1989(d 子项)",
),
)
register_step( register_step(
step_id="custom_value_check", step_id="custom_value_check",
title="自定义规则(字段值包含关键字)", title="自定义规则(字段值包含关键字)",
......
This diff is collapsed.
Markdown is supported
0%
or
You are about to add 0 people to the discussion. Proceed with caution.
Finish editing this message first!
Please register or to comment