Files
git-learn/skill-workbench/generated-skills/skills/value-scan/assets/check.py
T
zhuyongxin fddefab0c9 Add value-scan and value-dig skills for reverse-engineering value points
- value-scan: read-only breadth inventory of mechanisms in delivered code,

  stopping at the human selection gate

- value-dig: depth write-up of chosen points (feature list, design review,

  refactor plan) with templates and mechanical checkers

- Add skill-workbench design doc for the pair
2026-09-18 18:23:23 +08:00

308 lines
13 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
value-scan / value-dig 产物机械校验(Python 版,等价移植自 check.ps1,跨平台)
------------------------------------------------------------------
用法:
# S1 候选清单
python check.py scan <产物路径> --template <skills>/value-scan/assets/候选清单模板.md
# S4 功能点清单
python check.py dig <产物路径> --template <skills>/value-dig/assets/深度模板/功能点清单模板.md
# S4 设计思路与取舍 / 改造方案(整篇级文档,只跑通用校验)
python check.py doc <产物路径> --template <路径>
退出码:0 = 无 FAIL;1 = 有 FAIL;2 = 用法/IO 错误
说明:FAIL = 必错;WARN = 需人判断
"""
import argparse
import re
import sys
from pathlib import Path
FAIL, WARN, PASS = [], [], []
# ⟨ ⟩ 占位符(U+27E8 / U+27E9)
PH_L, PH_R = '\u27e8', '\u27e9'
def fail(m):
FAIL.append(m)
def warn(m):
WARN.append(m)
def pass_(m):
PASS.append(m)
def read_text(path_str):
p = Path(path_str)
if not p.is_file():
print(f"[ERROR] 文件不存在: {path_str}", file=sys.stderr)
sys.exit(2)
text = p.read_text(encoding='utf-8-sig') # 自动剥 BOM
return text, text.splitlines()
def head_index(lines, pattern):
for i, line in enumerate(lines):
if re.search(pattern, line):
return i
return -1
def next_head_index(lines, start):
for i in range(start + 1, len(lines)):
if re.match(r'^\s*#+\s+\S', lines[i]):
return i
return len(lines)
def line_snip(line, width=40):
"""报错附带的行内容摘要(改进:报错不指内容曾导致误诊)"""
s = line.strip().replace('|', '\\|')
return s[:width] + ('…' if len(s) > width else '')
def main():
ap = argparse.ArgumentParser(description='value-scan/value-dig 产物机械校验')
ap.add_argument('mode', choices=['scan', 'dig', 'doc'])
ap.add_argument('product', help='产物路径')
ap.add_argument('--template', '-t', help='模板路径')
ap.add_argument('-Mode', '-Product', '-Template', dest='legacy', help=argparse.SUPPRESS)
args = ap.parse_args()
prod_path = Path(args.product).resolve()
text, lines = read_text(args.product)
# ------------------------------------------------ 围栏代码块
fences = []
in_fence = False
f_start, f_lang = 0, ''
for i, line in enumerate(lines):
if re.match(r'^\s*```', line):
if not in_fence:
in_fence, f_start = True, i
f_lang = re.sub(r'^\s*```', '', line).strip()
else:
in_fence = False
fences.append({'start': f_start, 'end': i, 'lang': f_lang,
'body': i - f_start - 1})
if in_fence:
fail(f'存在未闭合的代码围栏(起始行 {f_start + 1})')
# ------------------------------------------------ 通用 1:引号逐行配对
odd_quote = [i + 1 for i, line in enumerate(lines)
if line.count('"') % 2 != 0]
if not odd_quote:
pass_('引号逐行配对')
else:
fail('引号未配对的行: ' + ', '.join(map(str, odd_quote)))
# ------------------------------------------------ 通用 2:mermaid 合法性
mermaid_fences = [f for f in fences if f['lang'] == 'mermaid']
mm_bad = 0
for f in mermaid_fences:
body_lines = lines[f['start'] + 1:f['end']]
body = '\n'.join(body_lines)
first = next((l for l in body_lines if l.strip()), '')
end_count = len(re.findall(r'(?m)^\s*end\s*$', body))
if 'sequenceDiagram' in first:
blk = len(re.findall(r'(?m)^\s*(alt|opt|loop|par|critical|break|rect)\b', body))
if blk != end_count:
fail(f"mermaid 时序图(第 {f['start'] + 1} 行起)alt/opt/loop 等 {blk} 个但 end={end_count}")
mm_bad += 1
else:
sg = len(re.findall(r'(?m)^\s*subgraph\s', body))
if sg != end_count:
fail(f"mermaid 流程图(第 {f['start'] + 1} 行起)subgraph={sg} 但 end={end_count}")
mm_bad += 1
if (re.search(r'(?m)^\s*subgraph\s+\S+\s*\[', body)
and not re.search(r'(?m)^\s*subgraph\s+\S+\s*\["', body)):
fail(f'mermaid(第 {f["start"] + 1} 行起)subgraph 缺引号标题,必须写 subgraph id["标题"]')
mm_bad += 1
if re.search(r'(?m)^\s*subgraph\s+\S+\s*\[\(', body):
fail(f'mermaid(第 {f["start"] + 1} 行起)用了 subgraph xxx[(...)],会解析失败')
mm_bad += 1
if mermaid_fences and mm_bad == 0:
pass_(f'mermaid 块 {len(mermaid_fences)} 个通过')
# ------------------------------------------------ 通用 3:表格列数一致(含行内容摘要)
ti = 0
while ti < len(lines):
if re.match(r'^\s*\|', lines[ti]):
bs = ti
blk = []
while ti < len(lines) and re.match(r'^\s*\|', lines[ti]):
blk.append(lines[ti])
ti += 1
counts = sorted({b.count('|') for b in blk})
if len(counts) > 1:
fail(f"表格列数不一致(第 {bs + 1} 行起):pipe 数 = {'/'.join(map(str, counts))}"
f"|首行内容: {line_snip(blk[0])}")
else:
ti += 1
# ------------------------------------------------ 通用 3b:表格不得缩进
indented = [i + 1 for i, line in enumerate(lines) if re.match(r'^[ \t]+\|', line)]
if not indented:
pass_('表格均顶格(无嵌套缩进表)')
else:
fail('表格存在缩进(第 ' + ', '.join(map(str, indented)) + ' 行起)——嵌套在列表里的表多数渲染器不显示,必须顶格')
# ------------------------------------------------ 通用 4:章节完整性
if args.template:
_, t_lines = read_text(args.template)
t_heads = [l for l in t_lines if re.match(r'^\s*#+\s+\S', l)]
missing_literal, missing_ph = [], []
for h in t_heads:
t = re.sub(r'\s+$', '', re.sub(r'^\s*#+\s+', '', h))
has_ph = PH_L in t
ph_rx = re.escape(PH_L) + '[^' + re.escape(PH_R) + ']*' + re.escape(PH_R)
rx = re.escape(re.sub(ph_rx, '@@PH@@', t)).replace('@@PH@@', '.*')
hit = any(re.match(r'^\s*#+\s+' + rx + r'\s*$', pl) for pl in lines)
if not hit:
(missing_ph if has_ph else missing_literal).append(t)
if not missing_literal:
pass_('章节完整性:模板必备节全部存在')
else:
fail('缺章节(字面量,必错): ' + ' | '.join(missing_literal))
if missing_ph:
warn('示例性章节未匹配(可能条数不同,需人判): ' + ' | '.join(missing_ph))
# ------------------------------------------------ scan 专属
if args.mode == 'scan':
non_mermaid = [f for f in fences if f['lang'] != 'mermaid']
if not non_mermaid:
pass_('深度闸门:S1 无围栏代码块(仅 mermaid)')
else:
fail('S1 不允许围栏代码块,发现 {} 个(第 {} 行起)'.format(
len(non_mermaid),
', '.join(str(f['start'] + 1) for f in non_mermaid)))
bad_anchor, anchor_count = [], 0
for line in lines:
m = re.match(r'^\s*\*\*锚点\*\*\s*[::]\s*(.+?)\s*$', line)
if not m:
continue
anchor_count += 1
items = [s for s in re.split(r'[、,,]', m.group(1)) if s.strip()]
if len(items) > 2:
bad_anchor.append(f'锚点超过 2 个({len(items)} 个): {line.strip()}')
for it in items:
a = it.strip().strip('`')
if re.search(r':\d', a):
bad_anchor.append(f'S1 锚点不写行号 -> {a}')
elif not re.search(r'\w[/\\]\w', a):
bad_anchor.append(f'S1 锚点须为 路径#方法名(机制级可到类名) -> {a}')
if anchor_count == 0:
warn('未找到 **锚点** 行(若产物为空则忽略)')
elif not bad_anchor:
pass_(f'锚点格式合规({anchor_count} 处,路径#方法名)')
else:
fail('锚点格式不合规: ' + ' ; '.join(bad_anchor))
if head_index(lines, r'^\s*#+\s*闸门状态') >= 0:
pass_('有「闸门状态」节(S2 门控有证据)')
else:
fail('缺「闸门状态」节 —— S2 门控没有证据')
star_count = sum(1 for l in lines if re.match(r'^\s*#+\s*★', l))
if star_count <= 12:
pass_(f'候选条数 {star_count}(≤12)')
else:
fail(f'候选条数 {star_count} 超过 12(颗粒度掉到实现层,需重并)')
if 0 < star_count < 5:
warn(f'候选条数 {star_count} 少于 5——先确认是否采样不到位,而非真没有')
# ------------------------------------------------ dig 专属
if args.mode == 'dig':
skel_idx = head_index(lines, r'^\s*#+\s*主干调用骨架')
skel_end = next_head_index(lines, skel_idx) if skel_idx >= 0 else -1
if skel_idx >= 0:
pass_('有「主干调用骨架」节')
else:
warn('未找到「主干调用骨架」节(仅 S4 ① 功能点清单必需)')
exempt = 0
for f in fences:
if f['lang'] == 'mermaid':
continue
if skel_idx >= 0 and skel_idx < f['start'] < skel_end:
exempt += 1
continue
if f['body'] > 10:
warn(f"局部代码块超过 10 行(第 {f['start'] + 1} 行起,{f['body']} 行)"
f"——行数非硬限,请人判断是关键片段还是源码摘录")
pass_(f'局部代码块行数检查完成(豁免骨架块 {exempt} 个)')
if skel_idx >= 0:
skel_fences = [f for f in fences if skel_idx < f['start'] < skel_end]
if not skel_fences:
warn('骨架节里没有代码块')
else:
sf = skel_fences[0]
miss_tag = []
for bl in lines[sf['start'] + 1:sf['end']]:
if not bl.strip():
continue
if re.match(r'^\s*//', bl):
continue
if re.search(r':\s*$', bl):
continue
if not re.search(r'[(\[]', bl):
continue
if '//' not in bl:
miss_tag.append(bl.strip())
if not miss_tag:
pass_('骨架每行都带方式标记')
else:
fail(f"骨架有 {len(miss_tag)} 行缺方式标记: "
+ ' ; '.join(miss_tag[:3]))
if not re.search(r'\.(java|xml|yaml|yml|sql):\d+', text):
fail('未找到任何 `文件:行号` 证据(本阶段必须有可核对的锚点)')
else:
pass_('存在 `文件:行号` 形式的证据')
# ------------------------------------------------ doc 专属
if args.mode == 'doc':
is_plan = head_index(lines, r'^\s*#+\s*0\.\s*现状问题登记') >= 0
if is_plan:
code_fences = [f for f in fences if f['lang'] != 'mermaid']
if len(code_fences) >= 1:
pass_(f'改造方案含关键片段 {len(code_fences)} 个')
else:
fail('改造方案缺「关键片段」代码块——必须"照着能讲代码"(伪代码/SQL 均可,只禁大段源码摘录)')
if len(code_fences) >= 2:
pass_(f"关键片段 {len(code_fences)} 个(骨架 + 改造点)")
elif len(code_fences) == 1:
warn('只有 1 个关键片段——可能只有链路骨架,各改造点未给伪代码,需人判')
# ------------------------------------------------ 输出
print()
print('=' * 68)
print(f'机械校验 Mode={args.mode}')
print(f' product : {prod_path}')
if args.template:
print(f" template : {Path(args.template).resolve()}")
print('=' * 68)
for tag, bucket, mark in (('[PASS]', PASS, '+ '), ('[WARN] 需人判断', WARN, '! '), ('[FAIL] 必错', FAIL, 'x ')):
if bucket:
print()
print(tag)
for m in bucket:
print(f' {mark}{m}')
print()
print(f'结果:PASS {len(PASS)} / WARN {len(WARN)} / FAIL {len(FAIL)}')
sys.exit(1 if FAIL else 0)
if __name__ == '__main__':
main()