Commit 9112c312 authored by Data Governance Dev's avatar Data Governance Dev

feat(step6): 字段名 + 字段注释 双路径匹配,结果带判断依据

原规则只按 column_name 正则匹配,对命名不规范但注释清楚的字段(如
code_a / value1 等通用名)漏判。本次扩展:

- 规则结构改 dataclass LengthRule:
    name_pattern: re.Pattern
    comment_keywords: tuple[str, ...]
    expected_length / standard / description
- 每条规则同时给出字段名正则 + 字段注释关键字列表(CI 子串匹配)
- 新增 _match_rule(col_name, col_comment, rule) → (hit, basis_text)
    字段名正则未命中再试注释关键字
- issue dict 新增字段 basis:
    "column_name 匹配 ^pattern$"  或  "column_comment 命中关键字「kw」"
- summary 新增 matched_by_name / matched_by_comment 计数

注释关键字示例(已覆盖常见中文命名习惯):
  身份证号 → 身份证 / 公民身份
  手机号   → 手机号 / 手机号码 / 移动电话
  信用代码 → 统一社会信用代码 / 社会信用代码 / 信用代码
  行政区划 → 行政区划 / 行政区划代码 / 地区编码
  邮政编码 → 邮政编码 / 邮编
  邮箱     → 邮箱 / 电子邮件 / e-mail
  银行卡   → 银行卡号 / 银行卡

前端 '字段长度异常' tab:'国家标准 / 规则' 列加宽 320→340,下方新增一行
'依据:<basis>' 小字斜体,便于审计。

烟雾测试 5 用例:t1.id_card(身份证号)、t4.region_code(空注释) 走字段名;
t2.code_a(注释含手机号)、t5.admin_code(注释含行政区划代码) 走注释关键字;
t3.remark(备注) 正确剔除。
parent 22ae5842
......@@ -3,41 +3,155 @@
识别字段定义长度超出标准所需(如身份证号 18 位用 varchar(50) 存储)。
仅基于内置规则匹配,不依赖 LLM。
匹配依据(双重判定):
1. 字段名正则:column_name 命中即匹配,依据记为 "column_name: <pattern>"
2. 字段注释关键字:column_comment 命中(不区分大小写、整词/子串均可)即匹配,
依据记为 "column_comment: <keyword>"
产出:
- summary: 总异常数
- issues: 每条具体异常(含浪费字节估算)
- issues: 每条具体异常(含浪费字节估算 + 匹配依据)
"""
from __future__ import annotations
import logging
import re
from dataclasses import dataclass
from typing import Callable
logger = logging.getLogger(__name__)
# 字段名模式 → 期望长度的规则
# 注意:期望长度需严格匹配国家标准
LENGTH_RULES: list[tuple[re.Pattern, int, str, str]] = [
# (pattern, expected_length, standard, description)
(re.compile(r"^(id_?card|id_?number|identity_?card)$", re.I), 18, "GB 11643-1999", "身份证号 18 位"),
(re.compile(r"^(id_?card_?no|_?sfz_?hm)$", re.I), 18, "GB 11643-1999", "身份证号 18 位"),
(re.compile(r"^(uscc|credit_?code|social_?credit_?code)$", re.I), 18, "GB 32100-2015", "统一社会信用代码 18 位"),
(re.compile(r"^mobile(_?phone)?$", re.I), 11, "YD/T 1313", "手机号 11 位"),
(re.compile(r"^phone(_?no)?$", re.I), 11, "YD/T 1313", "手机号 11 位"),
(re.compile(r"^tel(ephone)?$", re.I), 11, "YD/T 1313", "手机号 11 位"),
(re.compile(r"^(xzqhbm|adcode|district_?code)$", re.I), 6, "GB/T 2260", "行政区划代码 6 位"),
(re.compile(r"^province_?code$", re.I), 2, "GB/T 2260", "省级代码 2 位"),
(re.compile(r"^city_?code$", re.I), 4, "GB/T 2260", "市级代码 4 位"),
(re.compile(r"^region_?code$", re.I), 6, "GB/T 2260", "区县级代码 6 位"),
(re.compile(r"^zip_?code$", re.I), 6, "GB/T 23703", "邮政编码 6 位"),
(re.compile(r"^email$", re.I), 50, "RFC 5321", "电子邮件 ≤254 位,常用 ≤50"),
(re.compile(r"^bank_?card(_?no)?$", re.I), 19, "JR/T 0002", "银行卡号 ≤19 位"),
(re.compile(r"^post_?code$", re.I), 6, "GB/T 23703", "邮政编码 6 位"),
@dataclass
class LengthRule:
"""字段长度规则:同时支持字段名正则 + 字段注释关键字匹配。"""
name_pattern: re.Pattern # 字段名正则(命中即匹配)
comment_keywords: tuple[str, ...] # 字段注释关键字(任一命中即匹配,CI 子串匹配)
expected_length: int # 该类字段标准所需长度
standard: str # 依据标准编号
description: str # 规则描述
# 注意:comment_keywords 用小写存储;匹配时也对 column_comment 做 lower() 处理
LENGTH_RULES: list[LengthRule] = [
# ── 身份证号 (GB 11643-1999) ──
LengthRule(
name_pattern=re.compile(r"^(id_?card|id_?number|identity_?card)$", re.I),
comment_keywords=("身份证", "公民身份"),
expected_length=18, standard="GB 11643-1999",
description="身份证号 18 位",
),
LengthRule(
name_pattern=re.compile(r"^(id_?card_?no|_?sfz_?hm)$", re.I),
comment_keywords=("身份证",),
expected_length=18, standard="GB 11643-1999",
description="身份证号 18 位",
),
# ── 统一社会信用代码 (GB 32100-2015) ──
LengthRule(
name_pattern=re.compile(r"^(uscc|credit_?code|social_?credit_?code)$", re.I),
comment_keywords=("统一社会信用代码", "社会信用代码", "信用代码"),
expected_length=18, standard="GB 32100-2015",
description="统一社会信用代码 18 位",
),
# ── 手机号 (YD/T 1313) ──
LengthRule(
name_pattern=re.compile(r"^mobile(_?phone)?$", re.I),
comment_keywords=("手机号", "手机号码", "移动电话"),
expected_length=11, standard="YD/T 1313",
description="手机号 11 位",
),
LengthRule(
name_pattern=re.compile(r"^phone(_?no)?$", re.I),
comment_keywords=("联系电话",),
expected_length=11, standard="YD/T 1313",
description="手机号 11 位",
),
LengthRule(
name_pattern=re.compile(r"^tel(ephone)?$", re.I),
comment_keywords=("电话",),
expected_length=11, standard="YD/T 1313",
description="手机号 11 位",
),
# ── 行政区划代码 (GB/T 2260) ──
LengthRule(
name_pattern=re.compile(r"^(xzqhbm|adcode|district_?code)$", re.I),
comment_keywords=("行政区划", "行政区划代码", "地区编码"),
expected_length=6, standard="GB/T 2260",
description="行政区划代码 6 位",
),
LengthRule(
name_pattern=re.compile(r"^province_?code$", re.I),
comment_keywords=("省级代码", "省代码", "省份编码"),
expected_length=2, standard="GB/T 2260",
description="省级代码 2 位",
),
LengthRule(
name_pattern=re.compile(r"^city_?code$", re.I),
comment_keywords=("市级代码", "市代码", "城市编码"),
expected_length=4, standard="GB/T 2260",
description="市级代码 4 位",
),
LengthRule(
name_pattern=re.compile(r"^region_?code$", re.I),
comment_keywords=("区县级代码", "区县代码", "区县编码"),
expected_length=6, standard="GB/T 2260",
description="区县级代码 6 位",
),
# ── 邮政编码 (GB/T 23703) ──
LengthRule(
name_pattern=re.compile(r"^zip_?code$", re.I),
comment_keywords=("邮政编码", "邮编"),
expected_length=6, standard="GB/T 23703",
description="邮政编码 6 位",
),
LengthRule(
name_pattern=re.compile(r"^post_?code$", re.I),
comment_keywords=("邮政编码", "邮编"),
expected_length=6, standard="GB/T 23703",
description="邮政编码 6 位",
),
# ── 电子邮件 (RFC 5321) ──
LengthRule(
name_pattern=re.compile(r"^email$", re.I),
comment_keywords=("邮箱", "电子邮件", "e-mail"),
expected_length=50, standard="RFC 5321",
description="电子邮件 ≤254 位,常用 ≤50",
),
# ── 银行卡号 (JR/T 0002) ──
LengthRule(
name_pattern=re.compile(r"^bank_?card(_?no)?$", re.I),
comment_keywords=("银行卡号", "银行卡"),
expected_length=19, standard="JR/T 0002",
description="银行卡号 ≤19 位",
),
]
def _match_rule(col_name: str, col_comment: str, rule: LengthRule) -> tuple[bool, str]:
"""尝试匹配单条规则,返回 (是否命中, 命中依据文本).
字段名正则优先;未命中再试注释关键字。"""
# 1) 字段名正则
if rule.name_pattern.match(col_name):
return True, f"column_name 匹配 {rule.name_pattern.pattern}"
# 2) 字段注释关键字(不区分大小写、子串匹配)
if col_comment and rule.comment_keywords:
cc = col_comment.lower()
for kw in rule.comment_keywords:
if kw.lower() in cc:
return True, f"column_comment 命中关键字「{kw}」"
return False, ""
def run_step6(dict_data: dict, log: Callable | None = None) -> dict:
columns = dict_data.get("data_dictionary", [])
if not columns:
......@@ -51,44 +165,58 @@ def run_step6(dict_data: dict, log: Callable | None = None) -> dict:
step="6")
issues = []
rule_match_count = 0
match_by_name = 0
match_by_comment = 0
for r in columns:
col_name = r["column_name"]
col_comment = (r.get("column_comment") or "").strip()
dt = (r.get("data_type") or "").lower()
max_len = r.get("char_max_length")
if dt not in ("varchar", "char") or not max_len:
continue
for pat, expected, std, desc in LENGTH_RULES:
if pat.match(col_name):
rule_match_count += 1
if max_len > expected:
issues.append({
"table_name": r["table_name"],
"table_comment": r.get("table_comment", ""),
"column_name": col_name,
"data_type": dt,
"column_type": r.get("column_type", ""),
"actual_length": max_len,
"required_length": expected,
"wasted_bytes_per_row": max_len - expected,
"standard": std,
"rule": desc,
"issue": f"定义 {max_len} 位超出标准 {expected} 位",
"suggestion": f"改为 VARCHAR({expected})",
"column_comment": r.get("column_comment", ""),
})
break
for rule in LENGTH_RULES:
hit, basis = _match_rule(col_name, col_comment, rule)
if not hit:
continue
# 区分两种命中来源
if basis.startswith("column_name"):
match_by_name += 1
else:
match_by_comment += 1
if max_len > rule.expected_length:
issues.append({
"table_name": r["table_name"],
"table_comment": r.get("table_comment", ""),
"column_name": col_name,
"column_comment": col_comment,
"data_type": dt,
"column_type": r.get("column_type", ""),
"actual_length": max_len,
"required_length": rule.expected_length,
"wasted_bytes_per_row": max_len - rule.expected_length,
"standard": rule.standard,
"rule": rule.description,
"basis": basis, # ← 新增:判断依据
"issue": f"定义 {max_len} 位超出标准 {rule.expected_length} 位",
"suggestion": f"改为 VARCHAR({rule.expected_length})",
})
break # 同字段只取首个命中规则
if log:
log("INFO",
f"规则匹配 {rule_match_count} 次,发现异常 {len(issues)} 条",
f"规则匹配: 字段名 {match_by_name} 次 + 注释关键字 {match_by_comment} 次,"
f"发现异常 {len(issues)} 条",
step="6")
return {
"summary": {
"total_issues": len(issues),
"rules_applied": len(LENGTH_RULES),
"matched_by_name": match_by_name,
"matched_by_comment": match_by_comment,
"wasted_bytes_total": sum(i.get("wasted_bytes_per_row") or 0 for i in issues),
},
"issues": issues[:200],
......
......@@ -491,11 +491,12 @@
<el-table-column label="浪费字节" width="90" align="right">
<template #default="{ row }">{{ row.wasted_bytes_per_row }}</template>
</el-table-column>
<el-table-column label="国家标准 / 规则" width="320">
<el-table-column label="国家标准 / 规则 / 依据" width="340">
<template #default="{ row }">
<div>
<strong style="color: #303133">{{ row.standard }}</strong>
<div class="muted" style="font-size: 12px; margin-top: 2px">{{ row.rule }}</div>
<div v-if="row.basis" class="muted" style="font-size: 11px; margin-top: 2px; color: #909399; font-style: italic">依据:{{ row.basis }}</div>
</div>
</template>
</el-table-column>
......
Markdown is supported
0%
or
You are about to add 0 people to the discussion. Proceed with caution.
Finish editing this message first!
Please register or to comment