Commit 1bb3ca98 authored by Data Governance Dev's avatar Data Governance Dev

feat(standards): 细化违规原因 + 异常值字符位高亮 + tooltip 换行 + 合并 bug

合并两轮迭代的修改:

1) 违规原因细化(来自「细化所有目前显示的分析对象的错误信息」)
   - 21 个 IND validator + step4/5/6 的 reason 加上:
     · 标准来源编号(GB/T 2260 / GB 11643-1999 / 工信部 / GB 32100-2015 等)
     · 常见错因 ①②③(末位错填 / 出生日期段 / 地址码舍 0 等)
     · 建议核对原始证件 / 用脚本批量重算
   - reason 字符串内含 \n 换行符(前后端一起生效)

2) 异常值字符位高亮(用户截图反馈「异常值要指出哪里错了」)
   - standards/base.py:ValidationResult 新增 bad_positions 字段
   - 11 个 validator populate bad_positions:
     · IND-001/002/003/004/005/006 按字段类型填具体错位
     · IND-016 银行卡 Luhn 错:暴力枚举找出「单字符修正」位
   - web/core/step_impl/step7_standards.py:_build_value_highlighted
     · 区间合并(重叠/相邻)+ HTML escape → <mark class='bad-pos'>...
   - web/static/index.html:
     · .bad-pos CSS(淡红 + 红字 + 下划线 + cursor:help)
     · kind='code' 渲染分支 v-html='row.value_highlighted'

3) tooltip 多行修复(用户截图反馈「[a] [b] [c] 三个子检查挤在一行」)
   - 上一轮 .el-popper { white-space: pre-line !important } 看着对实际不生效
     · Element Plus 2.x show-overflow-tooltip 内部 normalize 文本,外层 CSS 救不回来
   - 本轮改用自渲染路线:放弃 show-overflow-tooltip,用 <el-tooltip> + #content 槽
     · escapeTooltipHtml():HTML escape + \n → <br>
     · cell 内 text-overflow: ellipsis 保持单行布局
     · 没 \n 的列无感升级

4) IND-003 / IND-004 合并步骤 TypeError 修复(用户截图反馈「有报错」)
   - 报错:TypeError: 'NoneType' object is not iterable
   - 根因:_merge_violations_by_tuple 里的死循环
     for p in (r.get('value_highlighted') and [])
     · 当 value_highlighted 为 None/空串时短路返回 None,
       for p in None 抛错
     · IND-003-b / IND-004-b 不填 bad_positions 时必触发
   - 修复:直接删这段死代码(loop 里只有 pass,真正 fallback 在下面)

5) 其他
   - docs/WORKLOG.md:四段工作记录(细化 / 高亮 / tooltip / 合并 bug)
   - scripts/reformat_reasons.py:AST-based 批量加 \n 工具,留作下轮兜底

未提交产物(生成报告 + tmp 脚本,不入库):
  - data_dictionary/standard_fields_violations_report.md
  - scripts/_tmp_build_violations_md.py

踩坑见 WORKLOG:
  - .el-tooltip__popper vs .el-popper(Element Plus 2.14.3 选择器)
  - and [] 短路求值返回 None(应为 or [] 或干脆删除)
  - show-overflow-tooltip 不可控,normalize 文本外层 CSS 救不回来
parent 5db40025
This diff is collapsed.
"""在 IND 校验 / step4/5/6 的「违规原因」reason 字符串中插入 \\n 换行符。
策略:
- 只动 ValidationResult(...) 的第 4 个位置参数(reason),不影响注释 / docstring / 其他字符串
- 用 ast.parse 精确定位 reason 的源码区间,只改这段文本
- 改完用 ast.unparse 重新拼回去 —— Python 语法上仍合法
转换规则(仅在 reason 字符串内部应用):
1) 跨行:→ " + 空白 + f" → 在 " 前插 \n (f-string 隐式拼接点)
2) 行内:句号 + 常见/建议/⚠ → 在中间插 \n
本脚本幂等:再次运行不会重复插入。
用法:
python scripts/reformat_reasons.py
"""
from __future__ import annotations
import ast
import re
import sys
from pathlib import Path
IND_FILES = [
"standards/ind_001a_id_card_format.py",
"standards/ind_001b_id_card_checksum.py",
"standards/ind_001c_id_card_birthdate.py",
"standards/ind_001d_id_card_15_legacy.py",
"standards/ind_002a_uscc_format.py",
"standards/ind_002b_uscc_checksum.py",
"standards/ind_002c_uscc_legacy_compat.py",
"standards/ind_003a_mobile_format.py",
"standards/ind_003b_mobile_segment.py",
"standards/ind_004a_xzqh_format.py",
"standards/ind_004b_xzqh_exists.py",
"standards/ind_005a_fixed_phone_format.py",
"standards/ind_006b_postal_code_format.py",
"standards/ind_007a_fund_account_format.py",
"standards/ind_013a_unit_type_enum.py",
"standards/ind_013b_economic_type_enum.py",
"standards/ind_013c_industry_code_format.py",
"standards/ind_016b_bank_account_format.py",
"standards/ind_401_name_charset.py",
"standards/ind_402_name_length.py",
]
STEP_IMPL_FILES = [
"web/core/step_impl/step4_empty_fields.py",
"web/core/step_impl/step5_missing_comments.py",
"web/core/step_impl/step6_length_check.py",
]
# ── 仅在 reason 字符串内应用的转换规则 ──────────────────────────────
# 顺序:先跨行(→ " + f"),再行内(。 + 常见/建议/⚠)
REASON_TRANSFORMS = [
# 跨行:箭头 → 在 f-string 末尾,后接空白 + 下一段 f" 开头
# 形如:... → "\n f"常见原因...
(r'→ "(\s+f")', r'→ \\n"\1'),
(r'→"(\s+f")', r'→\\n"\1'),
# 行内:句号 + 常见原因/错例
(r'。常见原因', r'。\\n常见原因'),
(r'。常见错例', r'。\\n常见错例'),
(r'。常见问题', r'。\\n常见问题'),
# 行内:句号 / 分号 + 建议
(r'。建议', r'。\\n建议'),
(r';建议', r';\\n建议'),
# 行内:句号 / 分号 + ⚠ 警告
(r'。⚠', r'。\\n⚠'),
(r';⚠', r';\\n⚠'),
]
def transform_reason(reason_src: str) -> tuple[str, int]:
"""对单个 reason 的源码片段应用转换规则,返回 (新文本, 命中次数)."""
counts = {}
out = reason_src
for pat, repl in REASON_TRANSFORMS:
out, n = re.subn(pat, repl, out)
counts[pat] = n
return out, sum(counts.values())
def find_reason_ranges(source: str) -> list[tuple[int, int]]:
"""解析 source,找到所有 ValidationResult(...) 调用的第 4 个位置参数(reason)
的源码区间 (start, end) 返回。"""
tree = ast.parse(source)
ranges = []
def _get_arg_src(node: ast.AST) -> tuple[int, int] | None:
"""从 AST 节点得到它在源码中的 (start, end) 字节偏移."""
if not hasattr(node, "lineno") or node.lineno is None:
return None
lines = source.splitlines(keepends=True)
start_line = node.lineno - 1
start_col = node.col_offset
end_line = node.end_lineno - 1
end_col = node.end_col_offset
start = sum(len(l) for l in lines[:start_line]) + start_col
end = sum(len(l) for l in lines[:end_line]) + end_col
return start, end
for node in ast.walk(tree):
if not isinstance(node, ast.Call):
continue
# 匹配 ValidationResult(...)
func = node.func
if not (isinstance(func, ast.Name) and func.id == "ValidationResult"):
continue
if len(node.args) < 4:
continue
# 第 4 个位置参数 (index 3) 是 reason
reason_node = node.args[3]
# 必须是字符串字面量(普通 str / JoinedStr)
if not isinstance(reason_node, (ast.Constant, ast.JoinedStr)):
continue
rng = _get_arg_src(reason_node)
if rng is None:
continue
ranges.append(rng)
return ranges
def transform_source(source: str) -> tuple[str, int]:
"""对 source 中所有 ValidationResult reason 字符串应用转换."""
ranges = find_reason_ranges(source)
if not ranges:
return source, 0
# 按 start 倒序排,从后往前替换避免偏移错位
ranges.sort(key=lambda r: r[0], reverse=True)
total = 0
for start, end in ranges:
original = source[start:end]
new, n = transform_reason(original)
if n > 0 and new != original:
source = source[:start] + new + source[end:]
total += n
return source, total
def main():
if hasattr(sys.stdout, "reconfigure"):
sys.stdout.reconfigure(encoding="utf-8")
project_root = Path(__file__).resolve().parent.parent
all_files = IND_FILES + STEP_IMPL_FILES
grand_total = 0
print(f"{'File':<60} {'Insertions':>10}")
print("-" * 72)
for rel in all_files:
path = project_root / rel
if not path.exists():
print(f"{rel:<60} {'(missing)':>10}")
continue
original = path.read_text(encoding="utf-8")
updated, total = transform_source(original)
if total > 0:
path.write_text(updated, encoding="utf-8")
grand_total += total
print(f"{rel:<60} {total:>10}")
print("-" * 72)
print(f"{'TOTAL':<60} {grand_total:>10}")
if __name__ == "__main__":
main()
......@@ -16,6 +16,13 @@ class ValidationResult:
reason: str = ""
standard: str = ""
# 2026-08-14:错误字符高亮 —— 给出 sample_value 中「具体哪里错了」的字符下标区间。
# 形如 [(start, end), ...],半开区间 [start, end),与 Python 切片语义一致。
# 例:身份证 `284228021973030610` 出生日期段非法 → [(6, 14)]
# 前端拿到后会把对应字符包在 <mark> 里展示,比纯文字 reason 更直观。
# 取 None 或空列表表示「无法定位 / 不高亮」。
bad_positions: list[tuple[int, int]] | None = None
def to_dict(self) -> dict:
return {
"valid": self.valid,
......@@ -23,6 +30,8 @@ class ValidationResult:
"sample": self.sample_value,
"reason": self.reason,
"standard": self.standard,
# JSON 不能直接传 tuple,转成 list[list[int]]
"bad_positions": [list(p) for p in (self.bad_positions or [])],
}
......
......@@ -24,7 +24,39 @@ class IdCardFormatIndicator(BaseStandard):
if not value:
return ValidationResult(True, "", value, "空值跳过", self.standard_id)
if len(value) != 18:
return ValidationResult(False, "", value, f"长度 {len(value)} ≠ 18", self.standard_id)
return ValidationResult(
False, "", value,
f"[GB 11643-1999 身份证格式] 实际长度 {len(value)} 位(应为 18 位)→ \n"
f"常见原因:①录入时漏字/截断;②字段类型为定长 CHAR(15/17) 被截;\n"
f"③老 15 位证未升级(见 IND-001-d)。建议核对原始证件补足 18 位后回填。",
self.standard_id,
# 长度异常 → 高亮整个值(前端用红色 mark 包住整段,让用户一眼看到「这串不对」)
bad_positions=[(0, len(value))],
)
if not self._REGEX.match(value):
return ValidationResult(False, "", value, "不符合 6+8+3+1 字符结构", self.standard_id)
# 结构不符:精确定位每一段违规位置,按以下顺序逐个判断
positions = []
if not value[0].isdigit() or value[0] == "0":
positions.append((0, 1)) # 地址码首位 1-9
else:
positions.append((0, 6)) # 地址码 6 位(粗粒度高亮整段,让用户审视)
# 出生日期段 6-14:若非数字则高亮 8-14(年/月/日),便于定位是年还是月日错
if not value[6:14].isdigit():
positions.append((6, 14))
else:
positions.append((6, 14)) # 即便数字也高亮日期段(多半 4 位年份或月日超范围)
# 末位校验位 17
if not (value[17].isdigit() or value[17].upper() == "X"):
positions.append((17, 18))
else:
positions.append((17, 18)) # 末位看着对但前 17 位错也算违规,整体高亮末位
return ValidationResult(
False, "", value,
f"[GB 11643-1999 身份证格式] 结构错误({value!r})→ 应为「6 位地址码 + 8 位出生日期(YYYYMMDD) + 3 位顺序码 + 1 位校验位(X/数字)」。"
f"常见错例:①末位错填非 X/数字 → 校验位(见 IND-001-b);"
f"②出生日期段填了非日期字符串 → 见 IND-001-c;"
f"③地址码首位含 0 → 身份证首位须 1-9(行政区划不存在 0 开头)。建议核对原始证件。",
self.standard_id,
bad_positions=positions,
)
return ValidationResult(True, "", value, "", self.standard_id)
\ No newline at end of file
......@@ -25,15 +25,23 @@ class IdCardChecksumIndicator(BaseStandard):
if len(value) != 18:
return ValidationResult(
False, "", value,
"校验位要求 18 位(请先确认 IND-001-a 格式)",
f"[GB 11643-1999 身份证校验位 ISO 7064 MOD 11-2] 长度 {len(value)} 位无法校验末位 → \n"
f"应先通过 IND-001-a 格式(18 位)后再做校验位。建议先修复长度/格式再回测。",
self.standard_id,
bad_positions=[(0, len(value))],
)
expected = self._calc(value[:17])
if value[17].upper() != expected:
# 末位错 → 高亮位置 17(单字符)。如果用户想精确定位前 17 位哪一位错,
# 那是 LLM/审计级需求;本次只做「肉眼可见的末位标记」。
return ValidationResult(
False, "", value,
f"校验位错误,应为 {expected}",
f"[GB 11643-1999 身份证校验位 ISO 7064 MOD 11-2] 末位错误:观测 '{value[17]}',正确应为 '{expected}' → \n"
f"权重 [7,9,10,5,8,4,2,1,6,3,7,9,10,5,8,4,2] × 前 17 位累加 mod 11。"
f"常见原因:①手工录入末位笔误;②前 17 位某位错填(身份证号常整体打错);\n"
f"③老 15 位证拼凑。建议重新核对原始证件或用脚本批量重算校验位。",
self.standard_id,
bad_positions=[(17, 18)],
)
return ValidationResult(True, "", value, "", self.standard_id)
......
......@@ -24,16 +24,25 @@ class IdCardBirthdateIndicator(BaseStandard):
if len(value) != 18:
return ValidationResult(
False, "", value,
"出生日期要求 18 位完整号码",
f"[GB 11643-1999 身份证出生日期] 长度 {len(value)} 位无法读取 6-14 位日期段 → \n"
f"应先通过 IND-001-a 格式(18 位)后再做日期校验。",
self.standard_id,
bad_positions=[(0, len(value))],
)
try:
y, m, d = int(value[6:10]), int(value[10:12]), int(value[12:14])
date_str = value[6:14]
try:
date(y, m, d)
except ValueError:
# 2026-08-14:高亮位置 6-14(出生日期段),便于用户一眼看到「这里日期错」
return ValidationResult(
False, "", value,
f"出生日期 {value[6:14]} 不合法(含闰年 / 月日范围)",
f"[GB 11643-1999 身份证出生日期] 位 7-14({date_str})不构成合法日期 → \n"
f"实际 Y={y} M={m:02d} D={d:02d}(含闰年 / 月日范围)。"
f"常见原因:①录入时月份或日期超出范围(如 13 月 / 32 日 / 2-30);\n"
f"②非闰年填了 02-29;③日期段被填为 000000 或 999999 等异常值。\n"
f"建议核对原始证件的出生年月。",
self.standard_id,
bad_positions=[(6, 14)],
)
return ValidationResult(True, "", value, "", self.standard_id)
\ No newline at end of file
......@@ -30,8 +30,14 @@ class IdCard15LegacyIndicator(BaseStandard):
if len(value) == 15 and self._REGEX_15.match(value):
return ValidationResult(
False, "", value,
"15 位老证,建议升级为 18 位(GB 11643-1999)",
f"[GB 11643-1989 身份证 15 位老证] 实际 '{value}' → 自 1999-07-01 起停止签发 15 位证(GB 11643-1999 启用 18 位)。"
f"15 位结构:6 位地址 + 6 位 YYMMDD + 3 位顺序码;"
f"升级为 18 位规则:YY → 19YY(<80)或 20YY(≥80)+ 补校验位。\n"
f"⚠ 仅 warning(业务上可能为历史数据,不强拒)。\n"
f"建议:评估是否需升级;批量升级可在 SQL 端用 15→18 转换函数。",
self.standard_id,
# 整段都是 15 位 → 全部高亮(让用户知道整段都需要升级,不是局部错)
bad_positions=[(0, len(value))],
)
# 非 15 位不算违规 —— 交给 IND-001-a/b/c
return ValidationResult(True, "", value, "", self.standard_id)
\ No newline at end of file
......@@ -25,10 +25,44 @@ class UsccFormatIndicator(BaseStandard):
if not value:
return ValidationResult(True, "", value, "空值跳过", self.standard_id)
if len(value) != 18:
return ValidationResult(False, "", value, f"长度 {len(value)} ≠ 18", self.standard_id)
return ValidationResult(
False, "", value,
f"[GB 32100-2015 USCC 格式] 实际长度 {len(value)} 位(应为 18 位)→ \n"
f"常见原因:①录入时漏字/截断;②字段类型过短(如 CHAR(15));\n"
f"③混入了旧 9 位组织机构代码(见 IND-002-c 兼容转换)。\n"
f"建议核对营业执照原件后回填完整 18 位。",
self.standard_id,
bad_positions=[(0, len(value))],
)
bad = set(value) & self._FORBIDDEN
if bad:
return ValidationResult(False, "", value, f"包含禁用字符 {''.join(sorted(bad))}", self.standard_id)
# 2026-08-14:精确定位每个禁用字符所在的位置 → 高亮这些字符
positions = [(i, i + 1) for i, c in enumerate(value) if c in bad]
return ValidationResult(
False, "", value,
f"[GB 32100-2015 USCC 格式] 含禁用字符 {''.join(sorted(bad))!r} → \n"
f"GB 32100 字符集仅 31 字符(0-9 + A-Z 去掉 I/O/Z/S/V)。\n"
f"常见原因:①录入时误把 O 当 0、I 当 1;②字段值混入了全角字符。\n"
f"建议清洗为 ASCII 半角字母数字。",
self.standard_id,
bad_positions=positions,
)
if not self._REGEX.match(value):
return ValidationResult(False, "", value, "不符合 1+1+6+9+1 结构", self.standard_id)
# 结构不符:先精确定位每一段违规(首字符 / 区划段 / 末位),其余段高亮整段让用户审视
positions = []
if value[0] not in "123456789ABCDEFGHJKLMNPQRSTUVWXYZ":
positions.append((0, 1)) # 登记管理部门错(首字符)
else:
positions.append((0, 1))
positions.append((2, 8)) # 行政区划段(6 位)粗粒度高亮
positions.append((17, 18)) # 末位校验位
return ValidationResult(
False, "", value,
f"[GB 32100-2015 USCC 格式] 结构不符 → 应为「1 位登记管理部门 + 1 位机构类别 + "
f"6 位行政区划 + 9 位主体标识码 + 1 位校验位」。\n"
f"常见原因:①首字符错填(含 0/小写字母);②某段长度错误(行政区划应为 6 位)。\n"
f"建议核对登记管理部门代码 + 区划代码。",
self.standard_id,
bad_positions=positions,
)
return ValidationResult(True, "", value, "", self.standard_id)
\ No newline at end of file
......@@ -25,18 +25,31 @@ class UsccChecksumIndicator(BaseStandard):
if len(value) != 18:
return ValidationResult(
False, "", value,
"校验位要求 18 位(请先确认 IND-002-a 格式)",
f"[GB 32100-2015 USCC 校验位 MOD 31-3] 长度 {len(value)} 位无法校验末位 → \n"
f"应先通过 IND-002-a 格式(18 位)后再做校验位。",
self.standard_id,
bad_positions=[(0, len(value))],
)
try:
expected = self._calc(value[:17])
except ValueError as e:
return ValidationResult(False, "", value, str(e), self.standard_id)
# 字符集外 → 高亮前 17 位(让用户知道整段都有问题)
return ValidationResult(
False, "", value,
f"[GB 32100-2015 USCC 校验位] {e}(前 17 位含 GB 32100 字符集外的字符)→ \n"
f"字符集:31 字符(0-9 + A-Z 去掉 I/O/Z/S/V)。建议清洗或重新录入。",
self.standard_id,
bad_positions=[(0, 17)],
)
if value[17] != expected:
return ValidationResult(
False, "", value,
f"校验位错误,应为 {expected}",
f"[GB 32100-2015 USCC 校验位 MOD 31-3] 末位错误:观测 '{value[17]}',正确应为 '{expected}' → \n"
f"Σ Ci × 3^i (i=0..16) → (31 - total%31)%31 倒索引查 GB 32100 字符集。"
f"常见原因:①前 17 位任一位错填;②录入时拷贝丢了校验位;③用 GB/T 12403 老代码校验位代替。\n"
f"建议重新核对营业执照或重新生成。",
self.standard_id,
bad_positions=[(17, 18)],
)
return ValidationResult(True, "", value, "", self.standard_id)
......
......@@ -147,24 +147,32 @@ class UsccLegacyCompatIndicator(BaseStandard):
if v[0] not in _LEGACY_TO_DEPT_CAT:
return ValidationResult(
False, "", value,
f"老代码首位 {v[0]!r} 不在 GB/T 12405 机构类别枚举 (1/2/3/4/5/9)",
f"[GB 32100-2015 老代码兼容转换] 老代码首位 {v[0]!r} 不在 GB/T 12405 机构类别枚举 → \n"
f"合法值:1(机关) / 2(事业) / 3(企业) / 4(社团) / 5(其他) / 9(个体工商)。"
f"常见原因:①录入时首位错填;②混入其它 9 位编码(如旧工商注册号)。\n"
f"建议核对登记证原件。",
self.standard_id,
)
if not _legacy_mod11_check(v):
return ValidationResult(
False, "", value,
f"老代码 {v!r} 校验位错误(GB/T 12403-1990 MOD 11)",
f"[GB 32100-2015 老代码兼容转换] 老代码 {v!r} 校验位错误 → \n"
f"GB/T 12403-1990 算法:权重 [3,7,9,10,5,8,4,2] × 前 8 位 → mod 11 → 查表(0-9, X)。\n"
f"建议重核原始证件或重新生成 9 位代码。",
self.standard_id,
)
new_uscc = self._convert(v)
return ValidationResult(
True, "", value,
f"老代码格式正确;建议替换为新 USCC: {new_uscc}",
f"[GB 32100-2015 老代码兼容转换] 老代码 {v} 格式正确 → 建议替换为新 USCC: {new_uscc}。\n"
f"⚠ 转换规则(GB 32100-2015 附录 A.1):部门/类别按首位映射 + 区划默认填 '110000'(北京,无法追溯登记地时的兜底)→ 业务上建议人工核对登记地后再定新 USCC。",
self.standard_id,
)
return ValidationResult(
False, "", value,
f"长度 {len(v)} 既不是 9 位老代码也不是 18 位新 USCC",
f"[GB 32100-2015 老代码兼容转换] 长度 {len(v)} 既不是 9 位老代码(GB/T 12405-2008)也不是 18 位新 USCC → \n"
f"常见原因:①截断(录入时位数不足);②混入了其它编码(如旧工商注册号 15 位 / 税务登记号)。\n"
f"建议核对营业执照上的统一社会信用代码。",
self.standard_id,
)
\ No newline at end of file
......@@ -50,11 +50,35 @@ class MobileFormatIndicator(BaseStandard):
return ValidationResult(True, "", value, "空值跳过", self.standard_id)
v = self._clean(value)
if len(v) != 11:
return ValidationResult(False, "", value, f"清洗后长度 {len(v)} ≠ 11", self.standard_id)
# 长度错 → 高亮整段
return ValidationResult(
False, "", value,
f"[工信部 手机号格式] 清洗后长度 {len(v)} 位(应为 11 位,自动去 +86/86 前缀、空格、横线)→ \n"
f"原始值 '{value}'。常见原因:①录入时漏号;②固话(座机)/ 短号(5 位企业号)/ 400/800 客服号混入;\n"
f"③字段实际为多个号码拼接(如 '139...;138...')。建议核对原始号码后回填。",
self.standard_id,
bad_positions=[(0, len(v))],
)
# 「不可全 0」前置拦截:00000000000 / 00000123456 等号段残缺值,
# 业务上视为无效 —— 给更具体的 reason,方便定位数据源问题
if set(v) == {"0"}:
return ValidationResult(False, "", value, "全 0 值(00000000000),业务上无效", self.standard_id)
return ValidationResult(
False, "", value,
f"[工信部 手机号格式] 全 0 值 '{v}' → 业务上视为无效(号段残缺值 / 系统初始默认值 / 测试数据)。\n"
f"建议:①核对原始采集系统是否做了空值兜底(如缺失手机号填 0);\n"
f"②批量清洗为 NULL 或剔除。",
self.standard_id,
bad_positions=[(0, 11)],
)
if not self._REGEX.match(v):
return ValidationResult(False, "", value, "不符合 1[3-9]XXXXXXXXX 格式", self.standard_id)
# 头部 1[3-9] 不符 → 高亮前 2 位(让用户看到「开头错」)
return ValidationResult(
False, "", value,
f"[工信部 手机号格式] 不符合 '1[3-9]XXXXXXXXX' 格式 → 实际 '{v}'。\n"
f"前 2 位必须 1[3-9](13-19 号段),如 130-139 / 145-149 / 150-153 / 155-159 / 166 / 171-178 / 180-189 / 191-193 / 195-199。\n"
f"常见原因:①录入时首位错填;②混入物联网卡(14 开头 13 位)、卫星电话(1740 开头);\n"
f"③0086 国际区号未清洗。",
self.standard_id,
bad_positions=[(0, 2)],
)
return ValidationResult(True, "", value, "", self.standard_id)
\ No newline at end of file
......@@ -35,5 +35,12 @@ class MobileSegmentIndicator(BaseStandard):
elif v.startswith("86") and len(v) == 13:
v = v[2:]
if not self._DETAILED_REGEX.match(v):
return ValidationResult(False, "", value, "号段不在已知号段表内", self.standard_id)
return ValidationResult(
False, "", value,
f"[工信部 手机号号段] '{v}' 不在已知号段表 → 已发行号段:13X / 14[5-9] / 15[0-35-9] / 16[2567] / 17[0-8] / 18X / 19[0-35-9]。\n"
f"常见原因:①录入时第二位错填(罕见号段如 140/141/154/16[01234] 等未启用);\n"
f"②号码为携号转网前的老号段记录;③实际为固话或短号混入。\n"
f"⚠ 仅在 IND-003-a 通过后再跑此检查;号段表更新可能滞后,建议人工核对。",
self.standard_id,
)
return ValidationResult(True, "", value, "", self.standard_id)
\ No newline at end of file
......@@ -26,12 +26,24 @@ class XzqhFormatIndicator(BaseStandard):
if not value:
return ValidationResult(True, "", value, "空值跳过", self.standard_id)
if not (self._REGEX_6.match(value) or self._REGEX_12.match(value)):
return ValidationResult(False, "", value, "长度必须为 6 或 12 位数字", self.standard_id)
return ValidationResult(
False, "", value,
f"[GB/T 2260 行政区划代码格式] 长度 {len(value)} 位(应为 6 位标准版或 12 位扩展版)→ \n"
f"实际 '{value}'。常见原因:①录入时漏位(如只填 4 位市级代码);\n"
f"②混入行政区划简称(如 '北京');③字段实际为邮编(6 位但语义不同)。\n"
f"建议:核对民政部最新区划表。",
self.standard_id,
bad_positions=[(0, len(value))],
)
province = value[:2]
if province not in load_district_province_codes():
return ValidationResult(
False, "", value,
f"省级代码 {province} 不在 GB/T 2260 表内",
f"[GB/T 2260 行政区划代码格式] 省级代码 '{province}' 不在 GB/T 2260 表内 → \n"
f"省级代码应取 GB/T 2260 现行 34 个省级行政区划前 2 位(含港澳台 / 直辖市 / 自治区)。\n"
f"常见原因:①录入首位错填;②旧版区划代码(如 90-99 前缀为历史保留段,未启用);\n"
f"③混入其它编码(如行业代码、国标代码)。建议核对民政部发布版本。",
self.standard_id,
bad_positions=[(0, 2)],
)
return ValidationResult(True, "", value, "", self.standard_id)
\ No newline at end of file
......@@ -24,13 +24,18 @@ class XzqhExistsIndicator(BaseStandard):
if len(value) != 6:
return ValidationResult(
False, "", value,
"仅校验 6 位标准版;12 位扩展版交由 IND-004-a 兜底",
f"[GB/T 2260 行政区划存在性] 长度 {len(value)} 位(仅校验 6 位标准版)→ \n"
f"12 位扩展版交由 IND-004-a 做格式校验。本检查依赖 GB/T 2260-2007 行政区划表(共 3406 条),\n"
f"若使用了 12 位扩展版(街道级)需自建对照表。",
self.standard_id,
)
if value not in load_district_codes():
return ValidationResult(
False, "", value,
"6 位代码不在 GB/T 2260-2007 表内",
f"[GB/T 2260 行政区划存在性] '{value}' 不在 GB/T 2260-2007 表(共 3406 条)内 → \n"
f"常见原因:①录入时某位错填(县级/街道级区划精度高,差异大);\n"
f"②旧版区划代码(民政部会定期调整撤并区划);③数据源使用了过期版本表。\n"
f"建议:核对民政部最新 GB/T 2260 版本。",
self.standard_id,
)
return ValidationResult(True, "", value, "", self.standard_id)
\ No newline at end of file
......@@ -55,9 +55,16 @@ class FixedPhoneFormatIndicator(BaseStandard):
return ValidationResult(True, "", value, "空值跳过", self.standard_id)
v = self._clean(value)
if not self._REGEX.match(v):
# 区号首位必须 0 → 错误就高亮首字符
positions = [(0, 1)] if not v.startswith("0") else [(0, len(v))]
return ValidationResult(
False, "", value,
"不符合「0XX/0XXX - 7~8 位」固话格式",
f"[GB/T 15835 固定电话格式] '{v}' 不符合 '0XX/0XXX-7~8 位' 格式 → \n"
f"应:① 区号以 0 开头(3~4 位,如 010/021/0755/0835);\n"
f"② 号码 7~8 位;③ 区号与号码间允许 '-' 或空格分隔。\n"
f"常见原因:①混入了手机号(11 位 1 开头);②传真号位数异常;③录入带括号或分机号;\n"
f"④区号首位 1-9(无 0/1 开头固话)。⚠ 工信部未公开固话号段表,仅做格式校验。",
self.standard_id,
bad_positions=positions,
)
return ValidationResult(True, "", value, "", self.standard_id)
\ No newline at end of file
......@@ -32,9 +32,19 @@ class PostalCodeFormatIndicator(BaseStandard):
return ValidationResult(True, "", value, "空值跳过", self.standard_id)
v = str(value).strip()
if not self._REGEX.match(v):
# 邮编非 6 位数字 → 高亮非数字字符位置
positions = [(i, i + 1) for i, c in enumerate(v) if not c.isdigit()]
if not positions:
# 长度不足 → 高亮整段
positions = [(0, len(v))]
return ValidationResult(
False, "", value,
f"长度 {len(v)} 或格式不符合 6 位数字",
f"[GB/T 23705-2009 邮政编码格式] '{v}' 不符合 6 位数字 → 实际长度 {len(v)} 位。\n"
f"常见原因:①录入时漏位(如 5 位);②混入行政区划代码(也 6 位但语义不同);\n"
f"③字段实际为 6 位邮编前缀(如 '100000')。\n"
f"⚠ 仅校验格式,不查存在性(邮编变动频繁,且新区域可能尚未公开)。\n"
f"建议核对邮政 6 位邮编。",
self.standard_id,
bad_positions=positions,
)
return ValidationResult(True, "", value, "", self.standard_id)
\ No newline at end of file
......@@ -51,13 +51,19 @@ class FundAccountFormatIndicator(BaseStandard):
if not self._REGEX.match(v):
return ValidationResult(
False, "", value,
"非全数字字符(含字母 / 符号)",
f"[个人公积金账号 格式] '{v}' 含非数字字符(字母 / 符号 / 全角字符)→ \n"
f"字段匹配走注释(不查字段名),合法值仅 0-9 数字。\n"
f"常见原因:①录入时混入字母(如单位前缀 'GJJ-');②字段实际为单位账号(含连字符);\n"
f"③OCR 识别混入空格/换行。⚠ 各省口径不统一(北京/上海/广州 18 位、深圳 10 位、其他混合),不查存在性/校验位。",
self.standard_id,
)
if len(v) not in self._ALLOWED_LENS:
return ValidationResult(
False, "", value,
f"长度 {len(v)} ∉ {self._ALLOWED_LENS}(公积金账号通常 10 / 12 / 18 位)",
f"[个人公积金账号 格式] 长度 {len(v)} ∉ {list(self._ALLOWED_LENS)} → \n"
f"公积金账号通常 10 / 12 / 18 位(各省口径不一:北京/上海/广州 18 位;深圳 10 位早期 11 位;其他 10/12/18 混合)。\n"
f"常见原因:①录入时漏位或多位;②混入了单位账号(通常 9-12 位);\n"
f"③字段实际为证件号。建议核对当地公积金中心规范。",
self.standard_id,
)
return ValidationResult(True, "", value, "", self.standard_id)
\ No newline at end of file
......@@ -107,6 +107,12 @@ class UnitTypeEnumIndicator(BaseStandard):
return ValidationResult(True, "", value, "", self.standard_id)
return ValidationResult(
False, "", value,
f"单位类型 {v!r} 不在白名单(GB/T 12402 + 市场监管总局市场主体类型码)",
f"[GB/T 12402-2017 单位类型] '{v}' 不在白名单 → 接受 2 位 / 4 位数字码 + 文字形式。\n"
f"合法数字码:10/11/12/13/14/15(企业法人/有限责任/股份/个人独资/合伙/个体工商户);\n"
f"20/21/22/23/24(事业/机关/社团/民非/基金会);30(其他);99(其他)。\n"
f"4 位扩展:1000/1100/1110/1120/1190/1200/1210/1290/2000/2100/2200/2300/2900/3000/4000/5000/9000。\n"
f"常见文字:公司/有限责任公司/股份有限公司/事业单位/机关/个体工商户/合伙企业 等。\n"
f"常见原因:①混入行业分类代码;②业务上无法分类的值(如 '工厂');③录入业务简称。\n"
f"⚠ 业务上不可自定义(无法归入现有分类)。建议业务侧补全类型字典或人工分类。",
self.standard_id,
)
\ No newline at end of file
......@@ -101,6 +101,14 @@ class EconomicTypeEnumIndicator(BaseStandard):
return ValidationResult(True, "", value, "", self.standard_id)
return ValidationResult(
False, "", value,
f"经济类型 {v!r} 不在白名单(国统字〔2011〕86 号)",
f"[国统字〔2011〕86 号 经济类型] '{v}' 不在白名单 → 3 位数字码或文字形式。\n"
f"合法大类:100 内资 / 200 港澳台 / 300 外资;\n"
f"子项:120 集体 / 130 股份合作 / 140 联营 / 150 有限责任 / 160 股份 / 170 私营 / "
f"210 港澳合资 / 220 港澳合作 / 230 港澳独资 / 240 港澳台股份 / "
f"310 中外合资 / 320 中外合作 / 330 外资 / 340 外商股份 + 兜底(190/290/390)。\n"
f"常见文字:国有 / 集体 / 股份合作 / 联营 / 有限责任 / 股份制 / 私营 / 中外合资 / 外资 等。\n"
f"常见原因:①3 位码录入错位(如 111/199/400 不存在);\n"
f"②业务上无法分类的值(如 '工厂');③混入了单位类型代码(IND-013-a 是 2/4 位)。\n"
f"建议核对统计局 2017 版分类。",
self.standard_id,
)
\ No newline at end of file
......@@ -60,7 +60,11 @@ class IndustryCodeFormatIndicator(BaseStandard):
if not m:
return ValidationResult(
False, "", value,
f"行业代码 {v!r} 不符合「门类字母 A~T + 2~4 位数字」格式(GB/T 4754-2017)",
f"[GB/T 4754-2017 行业代码格式] '{v}' 不符合「门类字母 A~T + 2~4 位数字」格式 → \n"
f"结构:3 位大类 / 4 位中类 / 5 位小类(门类 20 个:ABCDEFGHIJKLMNOPQRST)。\n"
f"常见原因:①门类用小写字母(应为大写 A~T);\n"
f"②数字段错填(如 5 位小类后再多 1 位);③纯数字编码(缺门类字母);\n"
f"④混入了行政区划代码(6 位纯数字)。建议核对 GB/T 4754-2017 表。",
self.standard_id,
)
digits = m.group(2)
......@@ -68,7 +72,9 @@ class IndustryCodeFormatIndicator(BaseStandard):
if digits[:2] == "00":
return ValidationResult(
False, "", value,
f"行业代码 {v!r} 数字段以 00 开头(GB/T 4754 实际最小从 X01 起)",
f"[GB/T 4754-2017 行业代码格式] '{v}' 数字段以 00 开头 → GB/T 4754 实际最小从 X01 起,不存在 X00 大类。\n"
f"常见原因:①录入时漏位(如 A001 而非 A01);②默认初始值 0 填充;\n"
f"③混入其它编码(如行政区划后 4 位)。⚠ 本 indicator 仅校验格式,行业存在性待后续 IND-013-d 做精细化。",
self.standard_id,
)
return ValidationResult(True, "", value, "", self.standard_id)
\ No newline at end of file
......@@ -77,21 +77,63 @@ class BankAccountFormatIndicator(BaseStandard):
v = str(value).strip().replace(" ", "").replace("-", "")
if not v.isdigit():
# 高亮每个非数字字符的位置
positions = [(i, i + 1) for i, c in enumerate(v) if not c.isdigit()]
return ValidationResult(
False, "", value,
f"银行账号 {value!r} 含非数字字符(含空格/横线已清洗)",
f"[JR/T 0025-2004 银行卡号格式] '{value}' 含非数字字符(已去空白和横线)→ \n"
f"字段匹配走注释(不查字段名),合法值仅 0-9 数字。\n"
f"常见原因:①录入时混入字母;②OCR 识别错;③含 16 位卡号 + 4 位有效期拼接;\n"
f"④字段实际为含 IBAN / BIC 等银行标识的字符串。",
self.standard_id,
bad_positions=positions,
)
if not (_MIN_LEN <= len(v) <= _MAX_LEN):
return ValidationResult(
False, "", value,
f"银行账号长度 {len(v)} ∉ [{_MIN_LEN}, {_MAX_LEN}]",
f"[JR/T 0025-2004 银行卡号格式] 长度 {len(v)} ∉ [{_MIN_LEN}, {_MAX_LEN}] → \n"
f"实际 '{v}'。常见长度:信用卡 16 位;借记卡 16-19 位;对公账户 19 位常见;老账户 12-17 位。\n"
f"常见原因:①录入时漏位;②字段拼接了多段;③混入非账号字段。",
self.standard_id,
bad_positions=[(0, len(v))],
)
if _LUHN_APPLY_MIN <= len(v) <= _LUHN_APPLY_MAX and not _luhn_check(v):
# Luhn 错:定位每一位「使得 Luhn 通过需要的调整」(暴力尝试,O(n))
positions = _luhn_bad_positions(v)
return ValidationResult(
False, "", value,
f"银行账号 Luhn 校验失败({len(v)} 位,按 ISO/IEC 7812)",
f"[JR/T 0025-2004 银行卡号格式] {len(v)} 位 Luhn 校验失败(ISO/IEC 7812 / GB/T 14504)→ \n"
f"算法:右起第 2 位起 ×2,≥10 时两位相加;全部相加 mod 10 = 0 视为有效。\n"
f"常见原因:①录入时某位错填(卡号常整体打错);②非 Luhn 强制的银行内部账户;\n"
f"⚠ 部分老账户不强制 Luhn(仅 warning)。建议重新核对原始卡号或剔除脏数据。",
self.standard_id,
bad_positions=positions,
)
return ValidationResult(True, "", value, "", self.standard_id)
def _luhn_bad_positions(digits: str) -> list[tuple[int, int]]:
"""暴力枚举:尝试把每一位改成 0-9,看哪一位能让 Luhn 通过。
O(n × 10) 适合 12~22 位(≤ 220 次 Luhn 计算,可忽略)。
返回所有「单独调整 1 位就能让 Luhn 通过」的位位置(去重)。
都失败时(如多位都错)返回整段。
"""
def _ok(d: str) -> bool:
return _luhn_check(d) if d.isdigit() and len(d) >= 2 else False
hits: list[int] = []
for i in range(len(digits)):
for c in "0123456789":
if c == digits[i]:
continue
cand = digits[:i] + c + digits[i+1:]
if _ok(cand):
hits.append(i)
break
if not hits:
return [(0, len(digits))]
# 合并相邻位
if len(hits) == 1:
return [(hits[0], hits[0] + 1)]
# 多位都能修 → 高亮整段(提示用户整体审视)
return [(0, len(digits))]
\ No newline at end of file
......@@ -67,7 +67,11 @@ class NameCharsetIndicator(BaseStandard):
if not (has_han or has_letter):
return ValidationResult(
False, "", value,
"姓名不含任何汉字或字母(疑似纯数字 / 纯特殊符号)",
f"[姓名字符集] '{v}' 不含任何汉字或 ASCII 字母 → 疑似纯数字 / 纯特殊符号 / 纯空格 / Emoji。\n"
f"允许字符:①汉字(CJK 基本 0x4E00-0x9FA5 + 扩展 A 0x3400-0x4DBF);\n"
f"②维吾尔等少数民族「·」/ 全角点「.」/ 半角空格;③ ASCII 字母。\n"
f"常见原因:①录入业务编码当作姓名;②手机号 / 身份证号混入姓名字段;\n"
f"③字段实际为 nick / 昵称(含 Emoji)。建议核对原始数据源。",
self.standard_id,
)
......@@ -75,7 +79,11 @@ class NameCharsetIndicator(BaseStandard):
if bad:
return ValidationResult(
False, "", value,
f"姓名包含不规范字符 {''.join(bad[:5])!r}",
f"[姓名字符集] '{v}' 含 {len(bad)} 个不规范字符 {''.join(bad[:5])!r} → \n"
f"允许:汉字 / 维吾尔族「·」/ 全角点「.」/ 半角空格 / ASCII 字母。\n"
f"禁止:纯数字 / 纯特殊符号 / 控制字符 / Emoji / 标点符号(除「·」)。\n"
f"常见原因:①录入混入部门编号 / 工号;②OCR 识别错;③字段实际为别名/昵称。\n"
f"建议核对原始数据或清洗。",
self.standard_id,
)
return ValidationResult(True, "", value, "", self.standard_id)
\ No newline at end of file
......@@ -33,13 +33,18 @@ class NameLengthIndicator(BaseStandard):
if n < self._MIN:
return ValidationResult(
False, "", value,
f"姓名长度 {n} < {self._MIN}",
f"[姓名长度] '{v}' 长度 {n} < {self._MIN}(业务惯例要求 ≥2 字符)→ \n"
f"常见原因:①录入漏字(仅剩姓);②字段实际为姓名首字母/简称;\n"
f"③姓名为单字(罕见但少数民族有)。建议核对原始证件上的全名。",
self.standard_id,
)
if n > self._MAX:
return ValidationResult(
False, "", value,
f"姓名长度 {n} > {self._MAX}",
f"[姓名长度] '{v}' 长度 {n} > {self._MAX}(公安部姓名长度上限 20,留余少数民族长姓名)→ \n"
f"常见原因:①拼接了多余信息(如 '张三 (zhangsan@xx.com)');\n"
f"②OCR 识别重复;③字段实际为备注/昵称(少数民族姓名常见 13-15 字)。\n"
f"建议清洗为纯姓名。",
self.standard_id,
)
return ValidationResult(True, "", value, "", self.standard_id)
\ No newline at end of file
......@@ -313,11 +313,27 @@ def run_step4(cfg: DBConfig, log: Callable | None = None,
}
if empty_rate >= high_threshold:
field_record["level"] = "high"
# 2026-08-13:前端「错误信息」细化。空值率 ≥80% 视为高空(疑似废弃 / 未启用),
# ≥50% 视为中空(启用率较低)。理由文本给业务方 + 数据治理侧共同参考:
# - 数据治理:判断是否下线 / 拆分到子表
# - 业务方:核对采集覆盖率 / 业务必要性
field_record["reason"] = (
f"[高空字段] 空值率 {empty_rate*100:.1f}%({empty_count}/{row_count} 行空)→ \n"
f"阈值 ≥{high_threshold*100:.0f}%。字段基本未使用(疑似废弃 / 系统未启用 / 采集未完成)。\n"
f"建议:①核对原始采集是否已运行;②评估是否下线字段;\n"
f"③若保留,确认业务必要性 + 补全数据。"
)
high_count += 1
high_in_table.append(field_record)
high_fields_all.append(field_record)
elif empty_rate >= mid_threshold:
field_record["level"] = "mid"
field_record["reason"] = (
f"[中空字段] 空值率 {empty_rate*100:.1f}%({empty_count}/{row_count} 行空)→ \n"
f"阈值 ≥{mid_threshold*100:.0f}%。字段启用率较低。\n"
f"建议:①核对采集覆盖率;②评估业务必要性;\n"
f"③若可选填,确认 NOT NULL 设计是否合理。"
)
mid_count += 1
mid_in_table.append(field_record)
mid_fields_all.append(field_record)
......
......@@ -132,13 +132,25 @@ def run_step5(dict_data: dict,
if not _is_missing(comment):
continue
by_table[r["table_name"]] += 1
# 2026-08-13:前端「错误信息」细化 —— 给业务方 + DBA 看时需要明确:
# - 影响范围:缺注释字段直接影响数据字典可读性、新人 on boarding、字段语义一致性
# - 治理建议:业务侧补 COMMENT;DBA 端加 ALTER 脚本批量回填;接入自动化文档系统
nullable = r.get("is_nullable", "")
data_type = r.get("data_type", "")
missing.append({
"table_name": r["table_name"],
"table_comment": r.get("table_comment", ""),
"column_name": r["column_name"],
"data_type": r.get("data_type", ""),
"data_type": data_type,
"column_type": r.get("column_type", ""),
"is_nullable": r.get("is_nullable", ""),
"is_nullable": nullable,
"reason": (
f"[缺失注释] 字段 {r['column_name']}(类型 {data_type or '未知'},可空 {nullable or '未知'})"
f"无 COMMENT 注释 → 影响:①数据字典不可读;②新人难以理解字段语义;\n"
f"③自动化文档生成系统无法收录;④ETL/接口字段映射易出错。\n"
f"建议:①业务侧补全 COMMENT;②DBA 端 ALTER TABLE ... MODIFY COLUMN ... COMMENT '...';\n"
f"③接入元数据管理平台自动同步。"
),
})
if log:
......
......@@ -280,6 +280,21 @@ def run_step6(dict_data: dict, log: Callable | None = None,
match_by_comment += 1
if max_len > rule.expected_length:
wasted = max_len - rule.expected_length
# 2026-08-13:前端「错误信息」细化 —— 给出
# - 标准依据(GB/T 编号 + 描述)
# - 实际定义 vs 标准要求
# - 单行浪费字节 + 大致容量估算(按 100 万 / 1 亿 行推算)
# - 修复建议(具体 DDL 写法)
# - 已知风险(utf8mb4 下 CHAR vs VARCHAR 的尾部空格差异)
issue_text = (
f"[{rule.standard} {rule.description}] 实际定义 {max_len} 位 > 标准要求 {rule.expected_length} 位"
f"(依据:{basis})→ 单行浪费 {wasted} 字节。\n"
f"容量估算(utf8mb4):{wasted * 1_000_000 / 1024 / 1024:.1f} MB / 100 万行;"
f"{wasted * 100_000_000 / 1024 / 1024 / 1024:.2f} GB / 1 亿行。\n"
f"建议 DDL:ALTER TABLE `{r['table_name']}` MODIFY COLUMN `{col_name}` VARCHAR({rule.expected_length}) ...; \n"
f"⚠ utf8mb4 + CHAR 定长存储会填充空格(VARCHAR 不填充)。"
)
issues.append({
"table_name": r["table_name"],
"table_comment": r.get("table_comment", ""),
......@@ -289,11 +304,11 @@ def run_step6(dict_data: dict, log: Callable | None = None,
"column_type": r.get("column_type", ""),
"actual_length": max_len,
"required_length": rule.expected_length,
"wasted_bytes_per_row": max_len - rule.expected_length,
"wasted_bytes_per_row": wasted,
"standard": rule.standard,
"rule": rule.description,
"basis": basis, # ← 新增:判断依据
"issue": f"定义 {max_len} 位超出标准 {rule.expected_length} 位",
"issue": issue_text,
"suggestion": f"改为 VARCHAR({rule.expected_length})",
})
break # 同字段只取首个命中规则
......
......@@ -49,6 +49,66 @@ _GROUP_ORDER = {
}
# ── 错误字符高亮 ─────────────────────────────────────────
# 2026-08-14:用户在截图里看到 `284228021973030610` 这种值想知道「哪里错了」。
# Validator 返回 bad_positions=[(start, end), ...],本函数把它转成 HTML(<mark> 包起来),
# 让前端 v-html 渲染时用红色高亮具体的错误字符段。
import html as _html
def _build_value_highlighted(
value: str,
bad_positions: list[tuple[int, int]] | list[list[int]] | None,
) -> str | None:
"""把 bad_positions 区间渲染成高亮 HTML。
Args:
value: 原始 sample 值。
bad_positions: 半开区间列表 [(6, 14), ...],或 None。
Returns:
高亮后的 HTML 字符串;bad_positions 为空/None 时返回 None(前端 fallback 到纯文本)。
"""
if not value or not bad_positions:
return None
# JSON 反序列化后可能变成 list[list[int]],统一转 tuple
norm: list[tuple[int, int]] = []
for p in bad_positions:
if not p or len(p) < 2:
continue
s, e = int(p[0]), int(p[1])
if s < 0:
s = 0
if e > len(value):
e = len(value)
if e <= s:
continue
norm.append((s, e))
if not norm:
return None
# 合并重叠 / 相邻区间,保证连续段不重复嵌套
norm.sort()
merged: list[list[int]] = [list(norm[0])]
for s, e in norm[1:]:
if s <= merged[-1][1]: # 重叠或紧贴 → 合并
merged[-1][1] = max(merged[-1][1], e)
else:
merged.append([s, e])
# 拼接 HTML
parts: list[str] = []
cur = 0
for s, e in merged:
if s > cur:
parts.append(_html.escape(value[cur:s]))
parts.append(
f'<mark class="bad-pos" '
f'title="错误字符位 {s+1}-{e}">'
f'{_html.escape(value[s:e])}</mark>'
)
cur = e
if cur < len(value):
parts.append(_html.escape(value[cur:]))
return "".join(parts)
# ── Per-indicator 单 tab 协议模板 ────────────────────────
def _make_single_indicator_tab(
*, key: str, title: str,
......@@ -688,6 +748,11 @@ def _run_round1_one(
"value": str(val)[:50],
"reason": result.reason,
})
# 2026-08-14:高亮错误字符 → bad_positions → HTML
val_str = str(val)[:50]
highlighted = _build_value_highlighted(
val_str, result.bad_positions
)
bucket["violations"].append({
"table_name": table,
"table_comment": col_record.get("table_comment", ""),
......@@ -696,7 +761,9 @@ def _run_round1_one(
"rule_type": std_id,
"standard_name": std_instance.standard_name,
"column_type": col_record.get("column_type", ""),
"value": str(val)[:50],
"value": val_str,
# 新增:带 <mark> 高亮的 HTML;None → 前端用纯文本
"value_highlighted": highlighted,
"error": result.reason,
})
......@@ -1709,6 +1776,19 @@ def _merge_violations_by_tuple(violations: list[dict]) -> list[dict]:
for rt, err in zip(rule_types, errors)
if rt and err
)
# 2026-08-14:合并多条时 value_highlighted 保留策略
# —— 单条路径已带 value_highlighted,合并路径沿用首条;
# value_highlighted 是 HTML,无法重渲染;坏位置数据在合并后丢了,
# 简化处理:若首条没有,从后续补一个,保持高亮至少有一段在。
# 之前这里有一段死循环 `for p in (r.get("value_highlighted") and [])`,
# 当 value_highlighted 为 None/空串时短路返回 None,`for p in None` 抛
# `TypeError: 'NoneType' object is not iterable` —— IND-003/004 等
# sub-check 不一定都填 bad_positions 的场景必触发。直接删。
if not base.get("value_highlighted"):
for r in rows[1:]:
if r.get("value_highlighted"):
base["value_highlighted"] = r["value_highlighted"]
break
# 记录合并条数(前端可选展示,本次不用,留作扩展点)
base["merged_sub_count"] = len(rows)
out.append(base)
......
......@@ -236,6 +236,48 @@
.section-card.is-collapsed .cond-tree {
display: none;
}
/* 2026-08-14:违规原因 tooltip 多行展示
*
* 【踩坑】Element Plus 2.14.3 用 .el-popper(不是 .el-tooltip__popper)。
* 早期 v1 / 旧版有 el-tooltip__popper 类,2.x 起 popper 元素统一走
* <div class="el-popper is-dark">…</div>
* 所以选择器必须是 .el-popper;用 .el-tooltip__popper 一行也不会命中。
*
* 同时加 !important 兜底,防止 Element Plus 后续版本 / 用户主题覆盖。
* 后端 reason 里按语义插入 \n(→ / 常见原因 / 建议 / ⚠),pre-line
* 会保留 \n 渲染为换行、折叠连续空格(不影响单行 tooltip 的展示)。
*/
.el-popper,
.el-popper.is-dark {
white-space: pre-line !important;
max-width: 600px;
line-height: 1.6;
}
/* tooltip popper 容器本身(直接包住插槽内容的 div) */
.el-popper > div {
white-space: pre-line !important;
max-width: 600px;
}
/* 2026-08-14:错误字符高亮 —— <mark class="bad-pos">
* - 后端 validator 返回 bad_positions 区间 → step7 转成 <mark> 包裹的 HTML
* - 前端 v-html 渲染(仅这个字段是可信 HTML,其他仍走文本插值)
* - mark 自带黄色背景,我们改成淡红 + 红字 + 圆角,更醒目
*/
.bad-pos {
background: #fef0f0;
color: #f56c6c;
padding: 0 2px;
margin: 0 1px;
border-radius: 3px;
font-weight: 600;
border-bottom: 1.5px solid #f56c6c;
cursor: help;
}
.bad-pos:hover {
background: #fde2e2;
}
</style>
<!-- 前端依赖全部本地化(避免 CDN 被墙/慢) -->
<!-- Element Plus CSS -->
......@@ -1054,7 +1096,10 @@
{{ row[col.prop] }}
</template>
<template v-else-if="col.render && col.render.kind === 'code'" #default="{ row }">
<code style="background: #f5f7fa; padding: 1px 6px; border-radius: 3px; color: #e6a23c;">{{ row[col.prop] }}</code>
<!-- 2026-08-14:value_highlighted 是后端 pre-escaped + <mark> 包裹的 HTML,
优先用它渲染(错误字符红色高亮);fallback 到纯文本 prop -->
<code v-if="row.value_highlighted" style="background: #f5f7fa; padding: 1px 6px; border-radius: 3px; color: #e6a23c;" v-html="row.value_highlighted"></code>
<code v-else style="background: #f5f7fa; padding: 1px 6px; border-radius: 3px; color: #e6a23c;">{{ row[col.prop] }}</code>
</template>
<template v-else-if="col.render && col.render.kind === 'tag'" #default="{ row }">
<el-tag :type="((col.render.value_map || {})[row[col.prop]] || col.render.fallback || {}).type || 'info'" size="small" :effect="((col.render.value_map || {})[row[col.prop]] || col.render.fallback || {}).effect || 'plain'" :style="((col.render.value_map || {})[row[col.prop]] || col.render.fallback || {}).italic ? 'font-style: italic' : ''">
......@@ -1100,8 +1145,22 @@
<template v-else-if="col.render && col.render.kind === 'number'" #default="{ row }">
{{ col.render.format === 'thousand_sep' ? Number(row[col.prop] || 0).toLocaleString() : row[col.prop] }}
</template>
<!-- 2026-08-14:违规原因 tooltip 多行
Element Plus 的 show-overflow-tooltip 把
cell textContent 塞 popper 后用 white-space:
normal,不显示 \n。这里改用 el-tooltip 自渲染
槽位:#content 走 v-html 把 \n 转 <br>,cell
内的文本保持单行(不影响表格布局)。
没 show_overflow_tooltip 时仍走纯文本插值。 -->
<template v-else-if="!col.render || !col.render.kind" #default="{ row }">
{{ row[col.prop] }}
<el-tooltip v-if="col.show_overflow_tooltip && row[col.prop]"
placement="top" effect="dark" :show-after="100">
<template #content>
<div style="max-width: 600px; line-height: 1.6; text-align: left; white-space: normal;" v-html="escapeTooltipHtml(row[col.prop])"></div>
</template>
<span style="display: inline-block; max-width: 100%; overflow: hidden; text-overflow: ellipsis; white-space: nowrap;">{{ row[col.prop] }}</span>
</el-tooltip>
<template v-else>{{ row[col.prop] }}</template>
</template>
</el-table-column>
</el-table>
......@@ -2545,6 +2604,22 @@
}
}
// 2026-08-14:违规原因 tooltip 多行渲染 —— 前端自处理 \n
// Element Plus overflow tooltip 把 cell textContent 塞进 popper,
// popper 默认 white-space: normal,即使数据里有 \n 也不换行;
// CSS 改 .el-popper { white-space: pre-line !important } 在某些
// 浏览器 / 主题下会被覆盖。这里走「自渲染」路线:自己用 el-tooltip
// + #content 槽,把 \n 转成 <br> 后用 v-html,绕开 popper 的
// textContent / white-space 限制。
function escapeTooltipHtml(text) {
if (text === null || text === undefined) return '';
return String(text)
.replace(/&/g, '&amp;')
.replace(/</g, '&lt;')
.replace(/>/g, '&gt;')
.replace(/\n/g, '<br>');
}
onMounted(() => {
loadDbDefaults(); // 先拉默认值(覆盖 form 初始值)
loadSteps();
......@@ -2577,6 +2652,7 @@
overallViolationCount, flatResultRows,
resolveRows, renderSummary, resolveRenderPath, resolveRender,
tabStatus, tabBadgeType, tabBadgeText, tabBadgeShow,
escapeTooltipHtml,
testConnection, startJob, cancelJob, resetAndStart,
downloadReport,
loadStandards,
......
Markdown is supported
0%
or
You are about to add 0 people to the discussion. Proceed with caution.
Finish editing this message first!
Please register or to comment