#!/usr/bin/env python # -*- coding: utf-8 -*- """build_prd.py —— PRD 构建工具。 三个子命令: backfill 把 prd/modules/*.md 的正文回灌进主 PRD 第 4 章,并重算派生计数 verify 只读校验:图片、锚点、REQ 编号、TODO、计数一致性、模块文件 lint export 配图就地量化 + pandoc 出 prd/Continental-Retail-APP-PRD.docx 用仓库自带的虚拟环境跑(pandoc 由 pypandoc 自带,不必单独安装): .venv/Scripts/python.exe scripts/build_prd.py verify .venv/Scripts/python.exe scripts/build_prd.py backfill .venv/Scripts/python.exe scripts/build_prd.py export 设计约束(改动前先读 prd/modules/README.md 的「同步契约」): * 回灌范围**仅限**模块文件 `## 附:本模块归拢信息` 分界线以上、且在头部信息表之后 的正文区。头部横幅与信息表是模块文件自己的脚手架,不进主文件。 * 正文区里只许出现三种链接写法,回灌时机械地折成纯锚点: `#锚点` 原样 / `./NN-模块.md#锚点` → `#锚点` / `../Continental-Retail-APP-PRD.md#锚点` → `#锚点`。 指向模块文件自身 `附-N` 小节的链接**不允许**出现在正文区——主文件没有那些小节。 * 正文区禁止出现「本次拆分 / 回灌 / 主文件 / 本文件」这类过程措辞:主文件是对外交付件, 只能有 V1.0 → V1.1 的修订语。违反即 lint 失败,否则每次回灌都要人工重洗一遍措辞。 * 派生数据由脚本重算(附录 D.2、头部规模声明、10.2 图例计数、模块头部信息表); 附录 B 权限矩阵与 10.2 条目本身**不自动重写**,只报告差异,由人决定怎么并。 """ from __future__ import print_function import collections import difflib import io import os import re import sys import unicodedata ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) PRD_DIR = os.path.join(ROOT, "prd") PRD = os.path.join(PRD_DIR, "Continental-Retail-APP-PRD.md") MOD_DIR = os.path.join(PRD_DIR, "modules") DOCX = os.path.join(PRD_DIR, "Continental-Retail-APP-PRD.docx") REFERENCE_DOCX = os.path.join(ROOT, "scripts", "reference.docx") IMAGE_DIRS = ["app-design-images", "images", "mini-program-images"] DIVIDER = u"## 附:本模块归拢信息" CH4_H1 = u"# 4 用户需求" BAN_WORDS = re.compile(u"本次拆分|回灌|主文件|本文件") LINK_RE = re.compile(r"(? s and re.match(r"^# 5[ .]", l)) body = [] for mod in mods: body.extend(mod.spliced()) body.append(u"") new = lines[:first41] + body + lines[end:] changed = sum(1 for a, b in zip(lines[first41:end], new[first41:len(new) - (len(lines) - end)]) if a != b) print(u"第 4 章:%d 行 → %d 行(4.1 起至第 5 章前),逐模块拼接 %d 个" % ( end - first41, len(body), len(mods))) if changed or (end - first41) != len(body): print(u" 与回灌前有差异") text = "\n".join(new) text = sync_derived(text, mods) if dry: d = difflib.unified_diff(lines, text.split("\n"), u"回灌前", u"回灌后", n=0, lineterm="") out = [l for l in d if l[:1] in "+-" and l[:3] not in ("---", "+++")] print(u"\n--dry-run:未写入。差异行 %d" % len(out)) for l in out[:80]: print(u" " + l[:160]) if len(out) > 80: print(u" ...(另有 %d 行)" % (len(out) - 80)) return 0 write(PRD, text) sync_module_headers(mods, text) print(u"已写入 %s" % os.path.relpath(PRD, ROOT)) print() return cmd_verify(argv) def sync_derived(text, mods): """重算派生数据:附录 D.2、头部规模声明、10.2 图例计数。""" lines = text.split("\n") body_reqs, body_todos = scan_body(lines) # ---- 附录 D.2 ---- d2s = next(i for i, l in enumerate(lines) if l.startswith("## D.2")) d2e = next((i for i, l in enumerate(lines) if i > d2s and l.startswith("## ")), len(lines)) fixed = 0 for i in range(d2s, d2e): cells = [c.strip() for c in lines[i].strip().strip("|").split("|")] if len(cells) < 6 or cells[0].startswith("-"): continue code = re.sub(r"\*", "", cells[2]).strip() if code not in body_reqs: continue n, p = len(body_reqs[code]), len(body_todos.get(code, ())) want = [str(n), str(p), u"%d%%" % pct(n, p)] if [cells[3], cells[4], cells[5]] != want: cells[3], cells[4], cells[5] = want lines[i] = "| " + " | ".join(cells) + " |" fixed += 1 print(u"附录 D.2:%d 行按正文重算" % fixed) text = "\n".join(lines) # ---- 10.2 图例:新增 / 沿用 / 关闭 ---- active, closed, bold = scan_102(lines) total = len(active) carry = total - bold old = re.search(u"共 \\*\\*(\\d+) 条\\*\\*;其余 (\\d+) 条沿用 V1\\.0", text) if old and (int(old.group(1)), int(old.group(2))) != (bold, carry): print(u" 10.2 图例:%s/%s → %d/%d" % (old.group(1), old.group(2), bold, carry)) text = re.sub(u"共 \\*\\*\\d+ 条\\*\\*;其余 \\d+ 条沿用 V1\\.0", u"共 **%d 条**;其余 %d 条沿用 V1.0" % (bold, carry), text) text = text.replace(u"不计入 %d。" % total, u"不计入 %d。" % total) # ---- 头部规模声明与附录 C.3 首句(CLAUDE.md 要求的「计数联动」,三处一起改)---- nreq = len(set(REQ_ANY_RE.findall(text))) nimg = len(IMG_RE.findall(text)) ntab = count_c3_rows(lines) nch4 = len(IMG_RE.findall("\n".join(ch4_lines(lines)))) subs = [ (u"(\\d+) 条编号需求", u"%d 条编号需求" % nreq), (u"其中 (\\d+) 条待业务确认", u"其中 %d 条待业务确认" % total), (u"、(\\d+) 张图、", u"、%d 张图、" % nimg), (u"(\\d+) 张正文表格", u"%d 张正文表格" % ntab), (u"第 4 章的 (\\d+) 张图", u"第 4 章的 %d 张图" % nch4), (u"本文档共 \\*\\*(\\d+) 张图\\*\\*", u"本文档共 **%d 张图**" % nimg), (u"\\*\\*(\\d+) 张正文表格\\*\\*", u"**%d 张正文表格**" % ntab), ] out, nfix = [], 0 for line in text.split("\n"): if u"本文档规模" in line or u"本文档共 **" in line: for pat, rep in subs: new_line = re.sub(pat, rep, line) if new_line != line: nfix += 1 line = new_line out.append(line) text = "\n".join(out) print(u"规模:需求 %d 条(待确认 %d)、图 %d 张、正文表格 %d 张;声明改写 %d 处" % ( nreq, total, nimg, ntab, nfix)) return text def ch4_lines(lines): """第 4 章的行(4.1 起至第 5 章前)。""" s = next(i for i, l in enumerate(lines) if re.match(r"^## 4\.1 ", l)) e = next(i for i, l in enumerate(lines) if i > s and re.match(r"^# 5[ .]", l)) return lines[s:e] def count_c3_rows(lines): """正文表格数 = 附录 C.3 的数据行数(C.3 只收正文表格,不含附录与 10.2 自身的清单表)。""" s = next(i for i, l in enumerate(lines) if l.startswith("## C.3")) e = next((i for i, l in enumerate(lines) if i > s and l.startswith("# ")), len(lines)) return sum(1 for i in range(s, e) if lines[i].startswith("| ") and not re.match(r"^\| *-", lines[i]) and not lines[i].startswith(u"| 序")) def scan_body(lines): """全文的 REQ 定义与 TODO,按模块码归类。""" reqs = collections.defaultdict(list) todos = collections.defaultdict(set) s102 = next(i for i, l in enumerate(lines) if l.startswith("## 10.2")) for i, line in enumerate(lines): if i >= s102: break m = REQ_DEF_RE.match(line) if m: reqs[m.group(2)].append(m.group(1)) for t in TODO_RE.findall(line): todos[t[4:7]].add(t) return reqs, todos def count_non_ch4_reqs(lines): """第 4 章以外(第 5–9 章、附录)定义的 REQ 数。""" s = lines.index(CH4_H1) e = next(i for i, l in enumerate(lines) if i > s and re.match(r"^# 5[ .]", l)) n = 0 s102 = next(i for i, l in enumerate(lines) if l.startswith("## 10.2")) for i, line in enumerate(lines): if s <= i < e or i >= s102: continue if REQ_DEF_RE.match(line): n += 1 return n def scan_102(lines): """10.2 登记表:返回(生效条目、已关闭条目、其中加粗的条数)。""" s = next(i for i, l in enumerate(lines) if l.startswith("## 10.2")) e = next(i for i, l in enumerate(lines) if i > s and l.startswith("## 10.3")) active, closed, bold = set(), set(), 0 for i in range(s, e): l = lines[i] if not l.startswith("| ") or "REQ-" in l[:2] or re.match(r"^\| *-", l): continue cell = l.strip().strip("|").split("|")[0].strip() m = re.search(r"REQ-[A-Z]{3}-\d{3}", cell) if not m: continue if cell.startswith("~~"): closed.add(m.group(0)) else: active.add(m.group(0)) if cell.startswith("**"): bold += 1 return active, closed, bold def sync_module_headers(mods, prd_text): """把模块头部信息表的「需求条数」一行按主文件重写——它是派生数据,最容易过期。""" lines = prd_text.split("\n") reqs, todos = scan_body(lines) n = 0 for mod in mods: code = mod.code if code not in reqs: continue total, pending = len(reqs[code]), len(todos.get(code, ())) want = u"| 需求条数 | %d(待确认 %d,完成度 %d%%) |" % (total, pending, pct(total, pending)) ml = mod.lines for i in range(mod.info_start, mod.info_end + 1): if ml[i].startswith(u"| 需求条数 |") and ml[i].rstrip() != want: ml[i] = want write(mod.path, "\n".join(ml)) n += 1 break print(u"模块头部信息表:%d 个的「需求条数」行已按主文件重写" % n) # ------------------------------------------------------------------ verify def cmd_verify(argv): text = read(PRD) lines = text.split("\n") bad = 0 banner = re.search(r"", text) body_text = text.replace(banner.group(0), "") if banner else text # 1 配图 imgs = [t for _, t in IMG_RE.findall(text)] miss = [p for p in imgs if not os.path.exists(os.path.join(PRD_DIR, p.replace("/", os.sep)))] print(u"1 配图引用 %d 处,缺失 %d" % (len(imgs), len(miss))) for p in miss[:10]: print(u" !! %s" % p) bad += len(miss) # 2 锚点与自包含 anchors = set() for l in lines: m = re.match(r"^(#{1,6}) +(.*)$", l) if m: anchors.add(slug(m.group(2).strip())) targets = [t for _, t in LINK_RE.findall(body_text)] inner = [t for t in targets if t.startswith("#")] outer = [t for t in targets if not t.startswith("#") and not t.startswith("http")] dead = sorted(set(t for t in inner if t[1:] not in anchors)) print(u"2 站内锚点 %d 条,失效 %d;正文外链文件 %d 条(自包含要求为 0)" % ( len(inner), len(dead), len(outer))) for d in dead[:15]: print(u" !! %s" % d) for o in outer[:5]: print(u" !! %s" % o) bad += len(dead) + len(outer) # 3 tbd(1.8 里那条禁令本身不算) tbd = [i + 1 for i, l in enumerate(lines) if re.search(r"\btbd\b", l, re.I) and u"不使用" not in l] print(u"3 tbd 出现 %d 处 %s" % (len(tbd), tbd[:5])) bad += len(tbd) # 4 REQ 编号。全文出现过的编号才是「需求条数」的口径; # 其中一部分(验收行、后台域、只以 TODO 形式存在的)没有加粗定义行,属正常。 ids = set(REQ_ANY_RE.findall(text)) defs = collections.Counter() for l in lines: m = REQ_DEF_RE.match(l) if m: defs[m.group(1)] += 1 redef = [k for k, v in defs.items() if v > 1] print(u"4 编号 %d 个(加粗定义行 %d,仅见于表格/TODO %d),重复定义 %d %s" % ( len(ids), len(defs), len(ids) - len(defs), len(redef), redef[:5])) bad += len(redef) # 5 TODO ↔ 10.2 双向 s102 = lines.index(next(l for l in lines if l.startswith("## 10.2"))) todos = set(TODO_RE.findall("\n".join(lines[:s102]))) active, closed, bold = scan_102(lines) orphan, ghost = sorted(todos - active - closed), sorted(active - todos) print(u"5 正文 TODO %d 个 / 10.2 生效 %d 条(另已关闭 %d,其中加粗 %d)" % ( len(todos), len(active), len(closed), bold)) if orphan: print(u" !! 正文有 TODO 但 10.2 未登记:%s" % orphan) if ghost: print(u" !! 10.2 登记但正文无 TODO:%s" % ghost) bad += len(orphan) + len(ghost) # 6 计数联动。表格数以附录 C.3 的口径为准(只收正文表格,不含附录与 10.2 自身的清单表) print(u"6 计数联动") c3rows = count_c3_rows(lines) print(u" 实际:需求 %d、待确认 %d、图 %d、正文表格 %d(附录 C.3 行数)" % ( len(ids), len(active), len(imgs), c3rows)) declared = [] for i, l in enumerate(lines): if u"本文档规模" in l or u"本文档共 **" in l: print(u" [%d] %s" % (i + 1, l.strip()[:150])) declared += [int(x) for x in re.findall(r"\d+", re.sub(r"^#+ |第 4 章的 \d+ 张图", "", l.strip()))] for label, actual in [(u"需求", len(ids)), (u"待确认", len(active)), (u"图", len(imgs)), (u"表", c3rows)]: if actual not in declared: print(u" !! 头部规模声明里找不到「%s = %d」" % (label, actual)) bad += 1 # 附录 C.2 逐行对配图(边界从 C.2 起,别把 C.1 的图源目录说明表算进来) c2 = next(i for i, l in enumerate(lines) if l.startswith("## C.2")) c3 = next(i for i, l in enumerate(lines) if i > c2 and l.startswith("## C.3")) c2paths = [] for i in range(c2, c3): if lines[i].startswith("| ") and not re.match(r"^\| *-", lines[i]): for _, t in IMG_RE.findall(lines[i]): c2paths.append(t) cells = [c.strip() for c in lines[i].strip().strip("|").split("|")] for c in cells: if re.search(r"\.(png|jpg|jpeg)$", c, re.I): c2paths.append(c.strip("`")) c2rows = sum(1 for i in range(c2, c3) if lines[i].startswith("| ") and not re.match(r"^\| *-", lines[i]) and not lines[i].startswith(u"| 序")) print(u" 附录 C.2 配图行 %d(应等于配图引用 %d)" % (c2rows, len(imgs))) if c2rows != len(imgs): bad += 1 # 7 模块文件 lint try: mods = load_modules() problems = lint_modules(mods) mimgs = sum(len(m.images()) for m in mods) print(u"7 模块文件 %d 个,正文区 lint %d 处问题;模块侧配图 %d 张" % ( len(mods), len(problems), mimgs)) for name, n, why, ctx in problems[:20]: print(u" !! %-26s %5d %s:%s" % (name, n, why, ctx)) bad += len(problems) except Exception as exc: # noqa: BLE001 print(u"7 模块文件解析失败:%s" % exc) bad += 1 print() print(u"=== 合计异常 %d 处 ===" % bad) return 1 if bad else 0 # ------------------------------------------------------------------ export def ensure_reference_docx(pypandoc): """没有 scripts/reference.docx 就生成一个:pandoc 默认模板 + 中日韩字体。 pandoc 默认模板的主题里 `` 是空的,中文字体由 Word 自行挑, 换台机器版式就变。这里把标题/正文的东亚字体钉死成 Windows 自带的两款。 已存在则原样使用——用户在 Word 里改过的版式不该被脚本覆盖。 """ if os.path.exists(REFERENCE_DOCX): return import subprocess import zipfile raw = subprocess.check_output( [pypandoc.get_pandoc_path(), "--print-default-data-file", "reference.docx"]) tmp = REFERENCE_DOCX + ".tmp" with open(tmp, "wb") as fh: fh.write(raw) src = zipfile.ZipFile(tmp) items = [(i, src.read(i.filename)) for i in src.infolist()] src.close() with zipfile.ZipFile(REFERENCE_DOCX, "w", zipfile.ZIP_DEFLATED) as out: for info, data in items: if info.filename == "word/theme/theme1.xml": x = data.decode("utf-8") # majorFont 管标题、minorFont 管正文,各自的第一个 就是东亚字体位 for tag, face in ((u"", u"微软雅黑"), (u"", u"等线")): k = x.index(tag) e = x.index(u'', k) x = x[:e] + u'' % face + x[e + len(u''):] data = x.encode("utf-8") out.writestr(info, data) os.remove(tmp) print(u"已生成 scripts/reference.docx(标题 微软雅黑 / 正文 等线)。" u"要改版式就用 Word 打开它直接调样式,脚本不会覆盖。") def cmd_export(argv): try: from PIL import Image except ImportError: print(u"缺 Pillow。用 .venv/Scripts/python.exe 跑,或 uv pip install pillow") return 1 try: import pypandoc except ImportError: print(u"缺 pypandoc。用 .venv/Scripts/python.exe 跑,或 uv pip install pypandoc-binary") return 1 # ---- 配图:调色板量化,就地覆盖 ---- # 不再另出一份 prd-export/ 副本——两套一模一样的图只是让仓库多背一份体积。 # 已经是 P 模式的说明压过了,跳过;日后新加的截图是 RGB/RGBA,export 时会自动压掉, # 免得仓库又慢慢长胖。原图在 git 历史里,要还原就 `git checkout -- prd/xxx.png`。 src_total = dst_total = n = skip = 0 for d in IMAGE_DIRS: for dirpath, _, files in os.walk(os.path.join(PRD_DIR, d)): for f in files: if not re.search(r"\.(png|jpg|jpeg)$", f, re.I): continue s = os.path.join(dirpath, f) im = Image.open(s) if im.mode == "P": skip += 1 continue before = os.path.getsize(s) if im.mode not in ("RGB", "L"): flat = Image.new("RGB", im.size, (255, 255, 255)) flat.paste(im, mask=im.split()[-1] if im.mode in ("RGBA", "LA") else None) im = flat im.quantize(colors=256).save(s, optimize=True) src_total += before dst_total += os.path.getsize(s) n += 1 if n: print(u"配图量化 %d 张:%.1f MB → %.1f MB(%d%%);已是调色板图跳过 %d 张" % ( n, src_total / 1048576.0, dst_total / 1048576.0, int(dst_total * 100.0 / src_total + 0.5) if src_total else 0, skip)) else: print(u"配图 %d 张均已量化,无需处理" % skip) # ---- docx ---- # 直接调 pandoc 而不用 pypandoc.convert_file:要拿 stderr 和退出码。 # 不传 --metadata=lang:zh-CN——pypandoc 捆的 pandoc 没带 translations/zh.yaml, # 传了只会每张表刷一行 "term Table has no translation" 警告,而 w:lang 一个字节也没变 # (中文的 eastAsia="zh-CN" 本来就在 reference.docx 的 styles.xml 里)。 import subprocess ensure_reference_docx(pypandoc) cmd = [pypandoc.get_pandoc_path(), PRD, "-o", DOCX, "--from=gfm", "--toc", "--toc-depth=3", "--resource-path=" + PRD_DIR, "--reference-doc=" + REFERENCE_DOCX] proc = subprocess.Popen(cmd, stderr=subprocess.PIPE) err = proc.communicate()[1].decode("utf-8", "replace") for l in err.split("\n"): if l.strip(): print(u" pandoc: " + l) if proc.returncode: print(u"pandoc 失败,退出码 %d" % proc.returncode) return 1 print(u"docx:%.1f MB → %s" % ( os.path.getsize(DOCX) / 1048576.0, os.path.relpath(DOCX, ROOT))) print(u"PDF 不由本脚本生成——pandoc 转 PDF 需要 LaTeX,本机没装;用 Word 打开 docx 另存为 PDF。") return 0 COMMANDS = {"backfill": cmd_backfill, "verify": cmd_verify, "export": cmd_export} def main(argv): if len(argv) < 2 or argv[1] not in COMMANDS: print(__doc__) return 2 return COMMANDS[argv[1]](argv[2:]) if __name__ == "__main__": sys.exit(main(sys.argv))