#!/usr/bin/env python3 # -*- coding: utf-8 -*- """ value-scan / value-dig 产物机械校验(Python 版,等价移植自 check.ps1,跨平台) ------------------------------------------------------------------ 用法: # S1 候选清单 python check.py scan <产物路径> --template /value-scan/assets/候选清单模板.md # S4 功能点清单 python check.py dig <产物路径> --template /value-dig/assets/深度模板/功能点清单模板.md # S4 设计思路与取舍 / 改造方案(整篇级文档,只跑通用校验) python check.py doc <产物路径> --template <路径> 退出码:0 = 无 FAIL;1 = 有 FAIL;2 = 用法/IO 错误 说明:FAIL = 必错;WARN = 需人判断 """ import argparse import re import sys from pathlib import Path FAIL, WARN, PASS = [], [], [] # ⟨ ⟩ 占位符(U+27E8 / U+27E9) PH_L, PH_R = '\u27e8', '\u27e9' def fail(m): FAIL.append(m) def warn(m): WARN.append(m) def pass_(m): PASS.append(m) def read_text(path_str): p = Path(path_str) if not p.is_file(): print(f"[ERROR] 文件不存在: {path_str}", file=sys.stderr) sys.exit(2) text = p.read_text(encoding='utf-8-sig') # 自动剥 BOM return text, text.splitlines() def head_index(lines, pattern): for i, line in enumerate(lines): if re.search(pattern, line): return i return -1 def next_head_index(lines, start): for i in range(start + 1, len(lines)): if re.match(r'^\s*#+\s+\S', lines[i]): return i return len(lines) def line_snip(line, width=40): """报错附带的行内容摘要(改进:报错不指内容曾导致误诊)""" s = line.strip().replace('|', '\\|') return s[:width] + ('…' if len(s) > width else '') def main(): ap = argparse.ArgumentParser(description='value-scan/value-dig 产物机械校验') ap.add_argument('mode', choices=['scan', 'dig', 'doc']) ap.add_argument('product', help='产物路径') ap.add_argument('--template', '-t', help='模板路径') ap.add_argument('-Mode', '-Product', '-Template', dest='legacy', help=argparse.SUPPRESS) args = ap.parse_args() prod_path = Path(args.product).resolve() text, lines = read_text(args.product) # ------------------------------------------------ 围栏代码块 fences = [] in_fence = False f_start, f_lang = 0, '' for i, line in enumerate(lines): if re.match(r'^\s*```', line): if not in_fence: in_fence, f_start = True, i f_lang = re.sub(r'^\s*```', '', line).strip() else: in_fence = False fences.append({'start': f_start, 'end': i, 'lang': f_lang, 'body': i - f_start - 1}) if in_fence: fail(f'存在未闭合的代码围栏(起始行 {f_start + 1})') # ------------------------------------------------ 通用 1:引号逐行配对 odd_quote = [i + 1 for i, line in enumerate(lines) if line.count('"') % 2 != 0] if not odd_quote: pass_('引号逐行配对') else: fail('引号未配对的行: ' + ', '.join(map(str, odd_quote))) # ------------------------------------------------ 通用 2:mermaid 合法性 mermaid_fences = [f for f in fences if f['lang'] == 'mermaid'] mm_bad = 0 for f in mermaid_fences: body_lines = lines[f['start'] + 1:f['end']] body = '\n'.join(body_lines) first = next((l for l in body_lines if l.strip()), '') end_count = len(re.findall(r'(?m)^\s*end\s*$', body)) if 'sequenceDiagram' in first: blk = len(re.findall(r'(?m)^\s*(alt|opt|loop|par|critical|break|rect)\b', body)) if blk != end_count: fail(f"mermaid 时序图(第 {f['start'] + 1} 行起)alt/opt/loop 等 {blk} 个但 end={end_count}") mm_bad += 1 else: sg = len(re.findall(r'(?m)^\s*subgraph\s', body)) if sg != end_count: fail(f"mermaid 流程图(第 {f['start'] + 1} 行起)subgraph={sg} 但 end={end_count}") mm_bad += 1 if (re.search(r'(?m)^\s*subgraph\s+\S+\s*\[', body) and not re.search(r'(?m)^\s*subgraph\s+\S+\s*\["', body)): fail(f'mermaid(第 {f["start"] + 1} 行起)subgraph 缺引号标题,必须写 subgraph id["标题"]') mm_bad += 1 if re.search(r'(?m)^\s*subgraph\s+\S+\s*\[\(', body): fail(f'mermaid(第 {f["start"] + 1} 行起)用了 subgraph xxx[(...)],会解析失败') mm_bad += 1 if mermaid_fences and mm_bad == 0: pass_(f'mermaid 块 {len(mermaid_fences)} 个通过') # ------------------------------------------------ 通用 3:表格列数一致(含行内容摘要) ti = 0 while ti < len(lines): if re.match(r'^\s*\|', lines[ti]): bs = ti blk = [] while ti < len(lines) and re.match(r'^\s*\|', lines[ti]): blk.append(lines[ti]) ti += 1 counts = sorted({b.count('|') for b in blk}) if len(counts) > 1: fail(f"表格列数不一致(第 {bs + 1} 行起):pipe 数 = {'/'.join(map(str, counts))}" f"|首行内容: {line_snip(blk[0])}") else: ti += 1 # ------------------------------------------------ 通用 3b:表格不得缩进 indented = [i + 1 for i, line in enumerate(lines) if re.match(r'^[ \t]+\|', line)] if not indented: pass_('表格均顶格(无嵌套缩进表)') else: fail('表格存在缩进(第 ' + ', '.join(map(str, indented)) + ' 行起)——嵌套在列表里的表多数渲染器不显示,必须顶格') # ------------------------------------------------ 通用 4:章节完整性 if args.template: _, t_lines = read_text(args.template) t_heads = [l for l in t_lines if re.match(r'^\s*#+\s+\S', l)] missing_literal, missing_ph = [], [] for h in t_heads: t = re.sub(r'\s+$', '', re.sub(r'^\s*#+\s+', '', h)) has_ph = PH_L in t ph_rx = re.escape(PH_L) + '[^' + re.escape(PH_R) + ']*' + re.escape(PH_R) rx = re.escape(re.sub(ph_rx, '@@PH@@', t)).replace('@@PH@@', '.*') hit = any(re.match(r'^\s*#+\s+' + rx + r'\s*$', pl) for pl in lines) if not hit: (missing_ph if has_ph else missing_literal).append(t) if not missing_literal: pass_('章节完整性:模板必备节全部存在') else: fail('缺章节(字面量,必错): ' + ' | '.join(missing_literal)) if missing_ph: warn('示例性章节未匹配(可能条数不同,需人判): ' + ' | '.join(missing_ph)) # ------------------------------------------------ scan 专属 if args.mode == 'scan': non_mermaid = [f for f in fences if f['lang'] != 'mermaid'] if not non_mermaid: pass_('深度闸门:S1 无围栏代码块(仅 mermaid)') else: fail('S1 不允许围栏代码块,发现 {} 个(第 {} 行起)'.format( len(non_mermaid), ', '.join(str(f['start'] + 1) for f in non_mermaid))) bad_anchor, anchor_count = [], 0 for line in lines: m = re.match(r'^\s*\*\*锚点\*\*\s*[::]\s*(.+?)\s*$', line) if not m: continue anchor_count += 1 items = [s for s in re.split(r'[、,,]', m.group(1)) if s.strip()] if len(items) > 2: bad_anchor.append(f'锚点超过 2 个({len(items)} 个): {line.strip()}') for it in items: a = it.strip().strip('`') if re.search(r':\d', a): bad_anchor.append(f'S1 锚点不写行号 -> {a}') elif not re.search(r'\w[/\\]\w', a): bad_anchor.append(f'S1 锚点须为 路径#方法名(机制级可到类名) -> {a}') if anchor_count == 0: warn('未找到 **锚点** 行(若产物为空则忽略)') elif not bad_anchor: pass_(f'锚点格式合规({anchor_count} 处,路径#方法名)') else: fail('锚点格式不合规: ' + ' ; '.join(bad_anchor)) if head_index(lines, r'^\s*#+\s*闸门状态') >= 0: pass_('有「闸门状态」节(S2 门控有证据)') else: fail('缺「闸门状态」节 —— S2 门控没有证据') star_count = sum(1 for l in lines if re.match(r'^\s*#+\s*★', l)) if star_count <= 12: pass_(f'候选条数 {star_count}(≤12)') else: fail(f'候选条数 {star_count} 超过 12(颗粒度掉到实现层,需重并)') if 0 < star_count < 5: warn(f'候选条数 {star_count} 少于 5——先确认是否采样不到位,而非真没有') # ------------------------------------------------ dig 专属 if args.mode == 'dig': skel_idx = head_index(lines, r'^\s*#+\s*主干调用骨架') skel_end = next_head_index(lines, skel_idx) if skel_idx >= 0 else -1 if skel_idx >= 0: pass_('有「主干调用骨架」节') else: warn('未找到「主干调用骨架」节(仅 S4 ① 功能点清单必需)') exempt = 0 for f in fences: if f['lang'] == 'mermaid': continue if skel_idx >= 0 and skel_idx < f['start'] < skel_end: exempt += 1 continue if f['body'] > 10: warn(f"局部代码块超过 10 行(第 {f['start'] + 1} 行起,{f['body']} 行)" f"——行数非硬限,请人判断是关键片段还是源码摘录") pass_(f'局部代码块行数检查完成(豁免骨架块 {exempt} 个)') if skel_idx >= 0: skel_fences = [f for f in fences if skel_idx < f['start'] < skel_end] if not skel_fences: warn('骨架节里没有代码块') else: sf = skel_fences[0] miss_tag = [] for bl in lines[sf['start'] + 1:sf['end']]: if not bl.strip(): continue if re.match(r'^\s*//', bl): continue if re.search(r':\s*$', bl): continue if not re.search(r'[(\[]', bl): continue if '//' not in bl: miss_tag.append(bl.strip()) if not miss_tag: pass_('骨架每行都带方式标记') else: fail(f"骨架有 {len(miss_tag)} 行缺方式标记: " + ' ; '.join(miss_tag[:3])) if not re.search(r'\.(java|xml|yaml|yml|sql):\d+', text): fail('未找到任何 `文件:行号` 证据(本阶段必须有可核对的锚点)') else: pass_('存在 `文件:行号` 形式的证据') # ------------------------------------------------ doc 专属 if args.mode == 'doc': is_plan = head_index(lines, r'^\s*#+\s*0\.\s*现状问题登记') >= 0 if is_plan: code_fences = [f for f in fences if f['lang'] != 'mermaid'] if len(code_fences) >= 1: pass_(f'改造方案含关键片段 {len(code_fences)} 个') else: fail('改造方案缺「关键片段」代码块——必须"照着能讲代码"(伪代码/SQL 均可,只禁大段源码摘录)') if len(code_fences) >= 2: pass_(f"关键片段 {len(code_fences)} 个(骨架 + 改造点)") elif len(code_fences) == 1: warn('只有 1 个关键片段——可能只有链路骨架,各改造点未给伪代码,需人判') # ------------------------------------------------ 输出 print() print('=' * 68) print(f'机械校验 Mode={args.mode}') print(f' product : {prod_path}') if args.template: print(f" template : {Path(args.template).resolve()}") print('=' * 68) for tag, bucket, mark in (('[PASS]', PASS, '+ '), ('[WARN] 需人判断', WARN, '! '), ('[FAIL] 必错', FAIL, 'x ')): if bucket: print() print(tag) for m in bucket: print(f' {mark}{m}') print() print(f'结果:PASS {len(PASS)} / WARN {len(WARN)} / FAIL {len(FAIL)}') sys.exit(1 if FAIL else 0) if __name__ == '__main__': main()