Add initial reference document in Word format with structured headings and content

This commit is contained in:
Guangfei.Zhao
2026-08-24 18:46:37 +08:00
parent 70c5cef4e1
commit d43d33c3cd
197 changed files with 9230 additions and 3754 deletions
+684
View File
@@ -0,0 +1,684 @@
#!/usr/bin/env python
# -*- coding: utf-8 -*-
"""build_prd.py —— PRD 构建工具。
三个子命令:
backfill 把 prd/modules/*.md 的正文回灌进主 PRD 第 4 章,并重算派生计数
verify 只读校验:图片、锚点、REQ 编号、TODO、计数一致性、模块文件 lint
export 配图就地量化 + pandoc 出 prd/Continental-Retail-APP-PRD.docx
用仓库自带的虚拟环境跑(pandoc 由 pypandoc 自带,不必单独安装):
.venv/Scripts/python.exe scripts/build_prd.py verify
.venv/Scripts/python.exe scripts/build_prd.py backfill
.venv/Scripts/python.exe scripts/build_prd.py export
设计约束(改动前先读 prd/modules/README.md 的「同步契约」):
* 回灌范围**仅限**模块文件 `## 附:本模块归拢信息` 分界线以上、且在头部信息表之后
的正文区。头部横幅与信息表是模块文件自己的脚手架,不进主文件。
* 正文区里只许出现三种链接写法,回灌时机械地折成纯锚点:
`#锚点` 原样 / `./NN-模块.md#锚点` → `#锚点` / `../Continental-Retail-APP-PRD.md#锚点` → `#锚点`。
指向模块文件自身 `附-N` 小节的链接**不允许**出现在正文区——主文件没有那些小节。
* 正文区禁止出现「本次拆分 / 回灌 / 主文件 / 本文件」这类过程措辞:主文件是对外交付件,
只能有 V1.0 → V1.1 的修订语。违反即 lint 失败,否则每次回灌都要人工重洗一遍措辞。
* 派生数据由脚本重算(附录 D.2、头部规模声明、10.2 图例计数、模块头部信息表);
附录 B 权限矩阵与 10.2 条目本身**不自动重写**,只报告差异,由人决定怎么并。
"""
from __future__ import print_function
import collections
import difflib
import io
import os
import re
import sys
import unicodedata
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
PRD_DIR = os.path.join(ROOT, "prd")
PRD = os.path.join(PRD_DIR, "Continental-Retail-APP-PRD.md")
MOD_DIR = os.path.join(PRD_DIR, "modules")
DOCX = os.path.join(PRD_DIR, "Continental-Retail-APP-PRD.docx")
REFERENCE_DOCX = os.path.join(ROOT, "scripts", "reference.docx")
IMAGE_DIRS = ["app-design-images", "images", "mini-program-images"]
DIVIDER = u"## 附:本模块归拢信息"
CH4_H1 = u"# 4 用户需求"
BAN_WORDS = re.compile(u"本次拆分|回灌|主文件|本文件")
LINK_RE = re.compile(r"(?<!\!)\[([^\]]*)\]\(([^)]+)\)")
IMG_RE = re.compile(r"!\[([^\]]*)\]\(([^)]+)\)")
# REQ 定义行:允许有序列表/无序列表前缀(4.8.6 用的是 `1. **REQ-MIN-007 ...**`
REQ_DEF_RE = re.compile(r"^(?:\d+\.\s+|[-*]\s+)?\*\*(REQ-([A-Z]{3})-\d{3})[ \*]")
REQ_ANY_RE = re.compile(r"REQ-[A-Z]{3}-\d{3}")
TODO_RE = re.compile(r"TODO\((REQ-[A-Z]{3}-\d{3})\)")
MOD_LINK_RE = re.compile(r"^\./(\d\d-[^#)]*\.md)(#.*)?$")
# 模块文件名 → 主文件里对应的小节锚点,由 load_modules() 填。
# 用于把 `[4.6](./06-PUR-采购.md)` 这种整文件链接折成 `[4.6](#46-采购)`。
MODULE_ANCHOR = {}
def read(path):
return io.open(path, encoding="utf-8").read()
def write(path, text):
io.open(path, "w", encoding="utf-8", newline="\n").write(text)
def slug(heading):
"""GitHub 风格锚点。先剥掉行内标记,再只保留字母数字与 - _ 空格。"""
h = re.sub(r"`([^`]*)`", r"\1", heading)
h = re.sub(r"\*\*?([^*]*)\*\*?", r"\1", h)
h = LINK_RE.sub(r"\1", h)
keep = "".join(c for c in h if unicodedata.category(c)[0] in "LN" or c in "-_ ")
return keep.lower().replace(" ", "-")
def pct(total, pending):
"""完成度:(需求 - 待确认) / 需求,四舍五入。
不能用内置 round()——它是银行家舍入,49.5 会进到 50 之外的方向,与文档里
人工算出来的数对不上。
"""
if not total:
return 0
return int((total - pending) * 100.0 / total + 0.5)
class Module(object):
"""一个模块文件。body 是要回灌的正文区,appendix 是分界线以下的归拢信息。"""
def __init__(self, path):
self.path = path
self.name = os.path.basename(path)
lines = read(path).split("\n")
self.lines = lines
m = re.match(r"^# (4\.\d+) (.*)$", lines[0])
if not m:
raise ValueError(u"%s 首行不是 `# 4.x 模块名`%r" % (self.name, lines[0]))
self.section, self.title = m.group(1), m.group(2).strip()
try:
self.divider = next(i for i, l in enumerate(lines) if l.startswith(DIVIDER))
except StopIteration:
raise ValueError(u"%s 找不到分界线 %s" % (self.name, DIVIDER))
# 头部信息表:紧跟横幅的 `| 项 | 值 |` 表
info = [i for i, l in enumerate(lines[:40]) if l.startswith("| ")]
if not info:
raise ValueError(u"%s 头部没有信息表" % self.name)
self.info_start, self.info_end = info[0], info[-1]
self.info = {}
for i in range(self.info_start, self.info_end + 1):
cells = [c.strip() for c in lines[i].strip().strip("|").split("|")]
if len(cells) == 2 and not cells[0].startswith("-"):
self.info[cells[0]] = cells[1]
self.code = self.info.get(u"模块码", "")
# 正文区起点:信息表之后,跳过可选的表说明行「模块概要」与可选的 `---` 分隔线。
# 两者都是模块文件自己的脚手架——14 个模块里 10 个有、4 个没有,不能当必选项。
i = self.info_end + 1
while i < self.divider and not lines[i].strip():
i += 1
if i < self.divider and lines[i].strip() == u"模块概要":
i += 1
while i < self.divider and not lines[i].strip():
i += 1
if i < self.divider and lines[i].strip() == "---":
i += 1
self.body_start = i
self.body = lines[self.body_start:self.divider]
# ---- 回灌 ----
def spliced(self):
"""把正文区转成主文件第 4 章里的样子:标题降一级、链接折成纯锚点、图片去掉 ../。"""
out = [u"## %s %s" % (self.section, self.title), u""]
body = list(self.body)
while body and not body[0].strip():
body.pop(0)
while body and not body[-1].strip():
body.pop()
for line in body:
out.append(self._transform(line))
return out
def _transform(self, line):
h = re.match(r"^(#{1,6}) ", line)
if h:
line = "#" + line # 模块 H1/H2 → 主文件 H2/H3
line = IMG_RE.sub(self._img, line)
line = LINK_RE.sub(self._link, line)
return line
@staticmethod
def _img(m):
tgt = m.group(2)
if tgt.startswith("../"):
tgt = tgt[3:] # ../app-design-images/x.png → app-design-images/x.png
return "![%s](%s)" % (m.group(1), tgt)
@staticmethod
def _link(m):
tgt = m.group(2)
if tgt.startswith("../Continental-Retail-APP-PRD.md#"):
tgt = tgt[tgt.index("#"):]
else:
mm = MOD_LINK_RE.match(tgt)
if mm:
# 带锚点的直接取锚点;整文件链接折成该模块的小节锚点
tgt = mm.group(2) or MODULE_ANCHOR.get(mm.group(1), tgt)
return "[%s](%s)" % (m.group(1), tgt)
# ---- 统计 ----
def reqs(self):
ids = []
for line in self.body:
m = REQ_DEF_RE.match(line)
if m:
ids.append(m.group(1))
return ids
def todos(self):
return set(TODO_RE.findall("\n".join(self.body)))
def images(self):
return [t for _, t in IMG_RE.findall("\n".join(self.body))]
def appendix_section(self, prefix):
"""取分界线以下某个 `### 附-N ...` 小节的行。"""
out, on = [], False
for line in self.lines[self.divider:]:
if line.startswith("### " + prefix):
on = True
continue
if on and line.startswith("### "):
break
if on:
out.append(line)
return out
def load_modules():
mods = []
for name in sorted(os.listdir(MOD_DIR)):
if name.endswith(".md") and name != "README.md":
mods.append(Module(os.path.join(MOD_DIR, name)))
mods.sort(key=lambda m: float(m.section[2:]))
MODULE_ANCHOR.clear()
for mod in mods:
MODULE_ANCHOR[mod.name] = "#" + slug(u"%s %s" % (mod.section, mod.title))
return mods
# ------------------------------------------------------------------ lint
def lint_modules(mods):
"""模块文件正文区的硬规则。违反就不该回灌——否则主文件会被过程措辞污染。"""
problems = []
for mod in mods:
for off, line in enumerate(mod.body):
n = mod.body_start + off + 1
bare = LINK_RE.sub(lambda m: m.group(1), line)
if BAN_WORDS.search(bare):
problems.append((mod.name, n, u"正文区出现过程措辞", line.strip()[:80]))
if "](#附-" in line:
problems.append((mod.name, n, u"链接指向模块自身的附-N 小节", line.strip()[:80]))
for _, tgt in LINK_RE.findall(line):
ok = (tgt.startswith("#")
or tgt.startswith("http")
or tgt.startswith("../Continental-Retail-APP-PRD.md#")
or MOD_LINK_RE.match(tgt))
if not ok:
problems.append((mod.name, n, u"不认识的链接写法", tgt[:80]))
mm = MOD_LINK_RE.match(tgt or "")
if mm and not mm.group(2) and mm.group(1) not in MODULE_ANCHOR:
problems.append((mod.name, n, u"整文件链接指向不存在的模块", tgt[:80]))
return problems
# ------------------------------------------------------------------ backfill
def cmd_backfill(argv):
dry = "--dry-run" in argv
mods = load_modules()
problems = lint_modules(mods)
if problems:
print(u"lint 失败,%d 处(回灌已中止):" % len(problems))
for name, n, why, ctx in problems[:40]:
print(u" %-28s %5d %s%s" % (name, n, why, ctx))
return 1
lines = read(PRD).split("\n")
s = lines.index(CH4_H1)
first41 = next(i for i, l in enumerate(lines) if re.match(r"^## 4\.1 ", l))
end = next(i for i, l in enumerate(lines) if i > s and re.match(r"^# 5[ .]", l))
body = []
for mod in mods:
body.extend(mod.spliced())
body.append(u"")
new = lines[:first41] + body + lines[end:]
changed = sum(1 for a, b in zip(lines[first41:end], new[first41:len(new) - (len(lines) - end)]) if a != b)
print(u"第 4 章:%d 行 → %d 行(4.1 起至第 5 章前),逐模块拼接 %d" % (
end - first41, len(body), len(mods)))
if changed or (end - first41) != len(body):
print(u" 与回灌前有差异")
text = "\n".join(new)
text = sync_derived(text, mods)
if dry:
d = difflib.unified_diff(lines, text.split("\n"), u"回灌前", u"回灌后", n=0, lineterm="")
out = [l for l in d if l[:1] in "+-" and l[:3] not in ("---", "+++")]
print(u"\n--dry-run:未写入。差异行 %d" % len(out))
for l in out[:80]:
print(u" " + l[:160])
if len(out) > 80:
print(u" ...(另有 %d 行)" % (len(out) - 80))
return 0
write(PRD, text)
sync_module_headers(mods, text)
print(u"已写入 %s" % os.path.relpath(PRD, ROOT))
print()
return cmd_verify(argv)
def sync_derived(text, mods):
"""重算派生数据:附录 D.2、头部规模声明、10.2 图例计数。"""
lines = text.split("\n")
body_reqs, body_todos = scan_body(lines)
# ---- 附录 D.2 ----
d2s = next(i for i, l in enumerate(lines) if l.startswith("## D.2"))
d2e = next((i for i, l in enumerate(lines) if i > d2s and l.startswith("## ")), len(lines))
fixed = 0
for i in range(d2s, d2e):
cells = [c.strip() for c in lines[i].strip().strip("|").split("|")]
if len(cells) < 6 or cells[0].startswith("-"):
continue
code = re.sub(r"\*", "", cells[2]).strip()
if code not in body_reqs:
continue
n, p = len(body_reqs[code]), len(body_todos.get(code, ()))
want = [str(n), str(p), u"%d%%" % pct(n, p)]
if [cells[3], cells[4], cells[5]] != want:
cells[3], cells[4], cells[5] = want
lines[i] = "| " + " | ".join(cells) + " |"
fixed += 1
print(u"附录 D.2%d 行按正文重算" % fixed)
text = "\n".join(lines)
# ---- 10.2 图例:新增 / 沿用 / 关闭 ----
active, closed, bold = scan_102(lines)
total = len(active)
carry = total - bold
old = re.search(u"\\*\\*(\\d+) 条\\*\\*;其余 (\\d+) 条沿用 V1\\.0", text)
if old and (int(old.group(1)), int(old.group(2))) != (bold, carry):
print(u" 10.2 图例:%s/%s%d/%d" % (old.group(1), old.group(2), bold, carry))
text = re.sub(u"\\*\\*\\d+ 条\\*\\*;其余 \\d+ 条沿用 V1\\.0",
u"共 **%d 条**;其余 %d 条沿用 V1.0" % (bold, carry), text)
text = text.replace(u"不计入 %d" % total, u"不计入 %d" % total)
# ---- 头部规模声明与附录 C.3 首句(CLAUDE.md 要求的「计数联动」,三处一起改)----
nreq = len(set(REQ_ANY_RE.findall(text)))
nimg = len(IMG_RE.findall(text))
ntab = count_c3_rows(lines)
nch4 = len(IMG_RE.findall("\n".join(ch4_lines(lines))))
subs = [
(u"(\\d+) 条编号需求", u"%d 条编号需求" % nreq),
(u"其中 (\\d+) 条待业务确认", u"其中 %d 条待业务确认" % total),
(u"、(\\d+) 张图、", u"%d 张图、" % nimg),
(u"(\\d+) 张正文表格", u"%d 张正文表格" % ntab),
(u"第 4 章的 (\\d+) 张图", u"第 4 章的 %d 张图" % nch4),
(u"本文档共 \\*\\*(\\d+) 张图\\*\\*", u"本文档共 **%d 张图**" % nimg),
(u"\\*\\*(\\d+) 张正文表格\\*\\*", u"**%d 张正文表格**" % ntab),
]
out, nfix = [], 0
for line in text.split("\n"):
if u"本文档规模" in line or u"本文档共 **" in line:
for pat, rep in subs:
new_line = re.sub(pat, rep, line)
if new_line != line:
nfix += 1
line = new_line
out.append(line)
text = "\n".join(out)
print(u"规模:需求 %d 条(待确认 %d)、图 %d 张、正文表格 %d 张;声明改写 %d" % (
nreq, total, nimg, ntab, nfix))
return text
def ch4_lines(lines):
"""第 4 章的行(4.1 起至第 5 章前)。"""
s = next(i for i, l in enumerate(lines) if re.match(r"^## 4\.1 ", l))
e = next(i for i, l in enumerate(lines) if i > s and re.match(r"^# 5[ .]", l))
return lines[s:e]
def count_c3_rows(lines):
"""正文表格数 = 附录 C.3 的数据行数(C.3 只收正文表格,不含附录与 10.2 自身的清单表)。"""
s = next(i for i, l in enumerate(lines) if l.startswith("## C.3"))
e = next((i for i, l in enumerate(lines) if i > s and l.startswith("# ")), len(lines))
return sum(1 for i in range(s, e)
if lines[i].startswith("| ") and not re.match(r"^\| *-", lines[i])
and not lines[i].startswith(u"| 序"))
def scan_body(lines):
"""全文的 REQ 定义与 TODO,按模块码归类。"""
reqs = collections.defaultdict(list)
todos = collections.defaultdict(set)
s102 = next(i for i, l in enumerate(lines) if l.startswith("## 10.2"))
for i, line in enumerate(lines):
if i >= s102:
break
m = REQ_DEF_RE.match(line)
if m:
reqs[m.group(2)].append(m.group(1))
for t in TODO_RE.findall(line):
todos[t[4:7]].add(t)
return reqs, todos
def count_non_ch4_reqs(lines):
"""第 4 章以外(第 5–9 章、附录)定义的 REQ 数。"""
s = lines.index(CH4_H1)
e = next(i for i, l in enumerate(lines) if i > s and re.match(r"^# 5[ .]", l))
n = 0
s102 = next(i for i, l in enumerate(lines) if l.startswith("## 10.2"))
for i, line in enumerate(lines):
if s <= i < e or i >= s102:
continue
if REQ_DEF_RE.match(line):
n += 1
return n
def scan_102(lines):
"""10.2 登记表:返回(生效条目、已关闭条目、其中加粗的条数)。"""
s = next(i for i, l in enumerate(lines) if l.startswith("## 10.2"))
e = next(i for i, l in enumerate(lines) if i > s and l.startswith("## 10.3"))
active, closed, bold = set(), set(), 0
for i in range(s, e):
l = lines[i]
if not l.startswith("| ") or "REQ-" in l[:2] or re.match(r"^\| *-", l):
continue
cell = l.strip().strip("|").split("|")[0].strip()
m = re.search(r"REQ-[A-Z]{3}-\d{3}", cell)
if not m:
continue
if cell.startswith("~~"):
closed.add(m.group(0))
else:
active.add(m.group(0))
if cell.startswith("**"):
bold += 1
return active, closed, bold
def sync_module_headers(mods, prd_text):
"""把模块头部信息表的「需求条数」一行按主文件重写——它是派生数据,最容易过期。"""
lines = prd_text.split("\n")
reqs, todos = scan_body(lines)
n = 0
for mod in mods:
code = mod.code
if code not in reqs:
continue
total, pending = len(reqs[code]), len(todos.get(code, ()))
want = u"| 需求条数 | %d(待确认 %d,完成度 %d%% |" % (total, pending, pct(total, pending))
ml = mod.lines
for i in range(mod.info_start, mod.info_end + 1):
if ml[i].startswith(u"| 需求条数 |") and ml[i].rstrip() != want:
ml[i] = want
write(mod.path, "\n".join(ml))
n += 1
break
print(u"模块头部信息表:%d 个的「需求条数」行已按主文件重写" % n)
# ------------------------------------------------------------------ verify
def cmd_verify(argv):
text = read(PRD)
lines = text.split("\n")
bad = 0
banner = re.search(r"<!--[\s\S]*?-->", text)
body_text = text.replace(banner.group(0), "") if banner else text
# 1 配图
imgs = [t for _, t in IMG_RE.findall(text)]
miss = [p for p in imgs if not os.path.exists(os.path.join(PRD_DIR, p.replace("/", os.sep)))]
print(u"1 配图引用 %d 处,缺失 %d" % (len(imgs), len(miss)))
for p in miss[:10]:
print(u" !! %s" % p)
bad += len(miss)
# 2 锚点与自包含
anchors = set()
for l in lines:
m = re.match(r"^(#{1,6}) +(.*)$", l)
if m:
anchors.add(slug(m.group(2).strip()))
targets = [t for _, t in LINK_RE.findall(body_text)]
inner = [t for t in targets if t.startswith("#")]
outer = [t for t in targets if not t.startswith("#") and not t.startswith("http")]
dead = sorted(set(t for t in inner if t[1:] not in anchors))
print(u"2 站内锚点 %d 条,失效 %d;正文外链文件 %d 条(自包含要求为 0" % (
len(inner), len(dead), len(outer)))
for d in dead[:15]:
print(u" !! %s" % d)
for o in outer[:5]:
print(u" !! %s" % o)
bad += len(dead) + len(outer)
# 3 tbd(1.8 里那条禁令本身不算)
tbd = [i + 1 for i, l in enumerate(lines)
if re.search(r"\btbd\b", l, re.I) and u"不使用" not in l]
print(u"3 tbd 出现 %d%s" % (len(tbd), tbd[:5]))
bad += len(tbd)
# 4 REQ 编号。全文出现过的编号才是「需求条数」的口径;
# 其中一部分(验收行、后台域、只以 TODO 形式存在的)没有加粗定义行,属正常。
ids = set(REQ_ANY_RE.findall(text))
defs = collections.Counter()
for l in lines:
m = REQ_DEF_RE.match(l)
if m:
defs[m.group(1)] += 1
redef = [k for k, v in defs.items() if v > 1]
print(u"4 编号 %d 个(加粗定义行 %d,仅见于表格/TODO %d),重复定义 %d %s" % (
len(ids), len(defs), len(ids) - len(defs), len(redef), redef[:5]))
bad += len(redef)
# 5 TODO ↔ 10.2 双向
s102 = lines.index(next(l for l in lines if l.startswith("## 10.2")))
todos = set(TODO_RE.findall("\n".join(lines[:s102])))
active, closed, bold = scan_102(lines)
orphan, ghost = sorted(todos - active - closed), sorted(active - todos)
print(u"5 正文 TODO %d 个 / 10.2 生效 %d 条(另已关闭 %d,其中加粗 %d" % (
len(todos), len(active), len(closed), bold))
if orphan:
print(u" !! 正文有 TODO 但 10.2 未登记:%s" % orphan)
if ghost:
print(u" !! 10.2 登记但正文无 TODO%s" % ghost)
bad += len(orphan) + len(ghost)
# 6 计数联动。表格数以附录 C.3 的口径为准(只收正文表格,不含附录与 10.2 自身的清单表)
print(u"6 计数联动")
c3rows = count_c3_rows(lines)
print(u" 实际:需求 %d、待确认 %d、图 %d、正文表格 %d(附录 C.3 行数)" % (
len(ids), len(active), len(imgs), c3rows))
declared = []
for i, l in enumerate(lines):
if u"本文档规模" in l or u"本文档共 **" in l:
print(u" [%d] %s" % (i + 1, l.strip()[:150]))
declared += [int(x) for x in re.findall(r"\d+", re.sub(r"^#+ |第 4 章的 \d+ 张图", "", l.strip()))]
for label, actual in [(u"需求", len(ids)), (u"待确认", len(active)),
(u"", len(imgs)), (u"", c3rows)]:
if actual not in declared:
print(u" !! 头部规模声明里找不到「%s = %d" % (label, actual))
bad += 1
# 附录 C.2 逐行对配图(边界从 C.2 起,别把 C.1 的图源目录说明表算进来)
c2 = next(i for i, l in enumerate(lines) if l.startswith("## C.2"))
c3 = next(i for i, l in enumerate(lines) if i > c2 and l.startswith("## C.3"))
c2paths = []
for i in range(c2, c3):
if lines[i].startswith("| ") and not re.match(r"^\| *-", lines[i]):
for _, t in IMG_RE.findall(lines[i]):
c2paths.append(t)
cells = [c.strip() for c in lines[i].strip().strip("|").split("|")]
for c in cells:
if re.search(r"\.(png|jpg|jpeg)$", c, re.I):
c2paths.append(c.strip("`"))
c2rows = sum(1 for i in range(c2, c3)
if lines[i].startswith("| ") and not re.match(r"^\| *-", lines[i])
and not lines[i].startswith(u"| 序"))
print(u" 附录 C.2 配图行 %d(应等于配图引用 %d" % (c2rows, len(imgs)))
if c2rows != len(imgs):
bad += 1
# 7 模块文件 lint
try:
mods = load_modules()
problems = lint_modules(mods)
mimgs = sum(len(m.images()) for m in mods)
print(u"7 模块文件 %d 个,正文区 lint %d 处问题;模块侧配图 %d" % (
len(mods), len(problems), mimgs))
for name, n, why, ctx in problems[:20]:
print(u" !! %-26s %5d %s%s" % (name, n, why, ctx))
bad += len(problems)
except Exception as exc: # noqa: BLE001
print(u"7 模块文件解析失败:%s" % exc)
bad += 1
print()
print(u"=== 合计异常 %d 处 ===" % bad)
return 1 if bad else 0
# ------------------------------------------------------------------ export
def ensure_reference_docx(pypandoc):
"""没有 scripts/reference.docx 就生成一个:pandoc 默认模板 + 中日韩字体。
pandoc 默认模板的主题里 `<a:ea typeface=""/>` 是空的,中文字体由 Word 自行挑,
换台机器版式就变。这里把标题/正文的东亚字体钉死成 Windows 自带的两款。
已存在则原样使用——用户在 Word 里改过的版式不该被脚本覆盖。
"""
if os.path.exists(REFERENCE_DOCX):
return
import subprocess
import zipfile
raw = subprocess.check_output(
[pypandoc.get_pandoc_path(), "--print-default-data-file", "reference.docx"])
tmp = REFERENCE_DOCX + ".tmp"
with open(tmp, "wb") as fh:
fh.write(raw)
src = zipfile.ZipFile(tmp)
items = [(i, src.read(i.filename)) for i in src.infolist()]
src.close()
with zipfile.ZipFile(REFERENCE_DOCX, "w", zipfile.ZIP_DEFLATED) as out:
for info, data in items:
if info.filename == "word/theme/theme1.xml":
x = data.decode("utf-8")
# majorFont 管标题、minorFont 管正文,各自的第一个 <a:ea/> 就是东亚字体位
for tag, face in ((u"<a:majorFont>", u"微软雅黑"), (u"<a:minorFont>", u"等线")):
k = x.index(tag)
e = x.index(u'<a:ea typeface=""/>', k)
x = x[:e] + u'<a:ea typeface="%s"/>' % face + x[e + len(u'<a:ea typeface=""/>'):]
data = x.encode("utf-8")
out.writestr(info, data)
os.remove(tmp)
print(u"已生成 scripts/reference.docx(标题 微软雅黑 / 正文 等线)。"
u"要改版式就用 Word 打开它直接调样式,脚本不会覆盖。")
def cmd_export(argv):
try:
from PIL import Image
except ImportError:
print(u"缺 Pillow。用 .venv/Scripts/python.exe 跑,或 uv pip install pillow")
return 1
try:
import pypandoc
except ImportError:
print(u"缺 pypandoc。用 .venv/Scripts/python.exe 跑,或 uv pip install pypandoc-binary")
return 1
# ---- 配图:调色板量化,就地覆盖 ----
# 不再另出一份 prd-export/ 副本——两套一模一样的图只是让仓库多背一份体积。
# 已经是 P 模式的说明压过了,跳过;日后新加的截图是 RGB/RGBA,export 时会自动压掉,
# 免得仓库又慢慢长胖。原图在 git 历史里,要还原就 `git checkout <commit> -- prd/xxx.png`。
src_total = dst_total = n = skip = 0
for d in IMAGE_DIRS:
for dirpath, _, files in os.walk(os.path.join(PRD_DIR, d)):
for f in files:
if not re.search(r"\.(png|jpg|jpeg)$", f, re.I):
continue
s = os.path.join(dirpath, f)
im = Image.open(s)
if im.mode == "P":
skip += 1
continue
before = os.path.getsize(s)
if im.mode not in ("RGB", "L"):
flat = Image.new("RGB", im.size, (255, 255, 255))
flat.paste(im, mask=im.split()[-1] if im.mode in ("RGBA", "LA") else None)
im = flat
im.quantize(colors=256).save(s, optimize=True)
src_total += before
dst_total += os.path.getsize(s)
n += 1
if n:
print(u"配图量化 %d 张:%.1f MB → %.1f MB%d%%);已是调色板图跳过 %d" % (
n, src_total / 1048576.0, dst_total / 1048576.0,
int(dst_total * 100.0 / src_total + 0.5) if src_total else 0, skip))
else:
print(u"配图 %d 张均已量化,无需处理" % skip)
# ---- docx ----
# 直接调 pandoc 而不用 pypandoc.convert_file:要拿 stderr 和退出码。
# 不传 --metadata=lang:zh-CN——pypandoc 捆的 pandoc 没带 translations/zh.yaml
# 传了只会每张表刷一行 "term Table has no translation" 警告,而 w:lang 一个字节也没变
# (中文的 eastAsia="zh-CN" 本来就在 reference.docx 的 styles.xml 里)。
import subprocess
ensure_reference_docx(pypandoc)
cmd = [pypandoc.get_pandoc_path(), PRD, "-o", DOCX,
"--from=gfm", "--toc", "--toc-depth=3",
"--resource-path=" + PRD_DIR, "--reference-doc=" + REFERENCE_DOCX]
proc = subprocess.Popen(cmd, stderr=subprocess.PIPE)
err = proc.communicate()[1].decode("utf-8", "replace")
for l in err.split("\n"):
if l.strip():
print(u" pandoc: " + l)
if proc.returncode:
print(u"pandoc 失败,退出码 %d" % proc.returncode)
return 1
print(u"docx%.1f MB → %s" % (
os.path.getsize(DOCX) / 1048576.0, os.path.relpath(DOCX, ROOT)))
print(u"PDF 不由本脚本生成——pandoc 转 PDF 需要 LaTeX,本机没装;用 Word 打开 docx 另存为 PDF。")
return 0
COMMANDS = {"backfill": cmd_backfill, "verify": cmd_verify, "export": cmd_export}
def main(argv):
if len(argv) < 2 or argv[1] not in COMMANDS:
print(__doc__)
return 2
return COMMANDS[argv[1]](argv[2:])
if __name__ == "__main__":
sys.exit(main(sys.argv))
Binary file not shown.